agent-flywheel 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. agent_flywheel/__init__.py +1067 -0
  2. agent_flywheel/_http.py +176 -0
  3. agent_flywheel/adapters/__init__.py +49 -0
  4. agent_flywheel/adapters/adk.py +268 -0
  5. agent_flywheel/adapters/langchain.py +288 -0
  6. agent_flywheel/adapters/prime_agent.py +1232 -0
  7. agent_flywheel/binding.py +360 -0
  8. agent_flywheel/child.py +63 -0
  9. agent_flywheel/cli.py +228 -0
  10. agent_flywheel/contract.py +411 -0
  11. agent_flywheel/endpoint.py +225 -0
  12. agent_flywheel/entrypoint.py +146 -0
  13. agent_flywheel/errors.py +87 -0
  14. agent_flywheel/event_stream.py +281 -0
  15. agent_flywheel/evidence.py +140 -0
  16. agent_flywheel/execution_identity.py +217 -0
  17. agent_flywheel/export.py +344 -0
  18. agent_flywheel/host_description.py +870 -0
  19. agent_flywheel/hosts/__init__.py +53 -0
  20. agent_flywheel/hosts/goose/plugin.json +7 -0
  21. agent_flywheel/hosts/scripts/percepteye_record.py +494 -0
  22. agent_flywheel/hosts/scripts/percepteye_report.py +192 -0
  23. agent_flywheel/install_hooks.py +301 -0
  24. agent_flywheel/introspection.py +619 -0
  25. agent_flywheel/isolation.py +227 -0
  26. agent_flywheel/outcomes.py +831 -0
  27. agent_flywheel/policy.py +490 -0
  28. agent_flywheel/production.py +1589 -0
  29. agent_flywheel/prompt.py +178 -0
  30. agent_flywheel/runner.py +1623 -0
  31. agent_flywheel/tracing.py +279 -0
  32. agent_flywheel/transport.py +328 -0
  33. agent_flywheel-0.1.0.dist-info/METADATA +429 -0
  34. agent_flywheel-0.1.0.dist-info/RECORD +37 -0
  35. agent_flywheel-0.1.0.dist-info/WHEEL +4 -0
  36. agent_flywheel-0.1.0.dist-info/entry_points.txt +2 -0
  37. agent_flywheel-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1067 @@
1
+ """agent-flywheel -- run your own agent against PerceptEye rollouts.
2
+
3
+ import agent_flywheel as fw
4
+
5
+ fw.serve(
6
+ "my_package.agent:build", # importable, so a child can rebuild it
7
+ agent_id="my-coding-agent",
8
+ input_shape="text",
9
+ control_plane_url="https://launch.percepteye.ai/api/flywheel/v1",
10
+ )
11
+
12
+ Outbound only. The flywheel dials out, claims a rollout, runs it, reports back.
13
+ No listening socket, no inbound firewall rule, no public DNS.
14
+
15
+ Two things are worth knowing before you read further.
16
+
17
+ **There are two modes, and they are a deployment switch.** ``training`` --
18
+ Percepteye supplies the model endpoint, your flywheel supplies the agent loop
19
+ -- and ``production``, in which ``serve()`` returns 0 without connecting to
20
+ anything. Set ``PERCEPTEYE_AGENT_MODE`` and the same image does either, because
21
+ "stop contributing capacity" is an operational decision and it should not require
22
+ editing the source to express it.
23
+
24
+ That is the ONLY axis this word names here. An earlier SDK shipped five modes
25
+ of which four were byte-identical strings in one payload field the server
26
+ never read, and the rule that came out of it stands: a mode has to change what
27
+ the SDK does. These two do -- one claims rollouts and the other opens no
28
+ socket.
29
+
30
+ On the wire, ``serve()`` registers as ``"rollout"`` -- that field names the
31
+ protocol this process speaks, and it is the value intake has always received
32
+ from a training donor. ``attach()`` registers as ``"production"``, because a
33
+ training donor and a production reader are the one distinction a ``mode`` on a
34
+ registration exists to draw, and while both sent ``"rollout"`` the field
35
+ answered nothing. So the wire carries two values and this parameter carries
36
+ two modes, but they are not the same axis and do not have to agree.
37
+
38
+ **Reporting "I don't know" is a supported answer.** When your tool wrapper
39
+ cannot tell whether a call succeeded, say ``outcome="unknown"``. It is
40
+ recorded as unknown and the verifiers that depend on it abstain. It is never
41
+ rounded up. This matters more than it sounds: the alternative is a reward
42
+ signal that scores a failed call as a landed one, and a policy trained to
43
+ believe it.
44
+ """
45
+ from __future__ import annotations
46
+
47
+ import logging
48
+ import os
49
+ import warnings
50
+ from collections.abc import Callable, Mapping
51
+ from typing import Any, Literal
52
+
53
+ from .contract import (
54
+ CONTRACT_VERSION,
55
+ ExclusionReason,
56
+ Outcome,
57
+ RolloutOutput,
58
+ RolloutRequest,
59
+ ToolCallOutcome,
60
+ )
61
+ from .entrypoint import TASK, InputShape, resolve_entrypoint
62
+ from .errors import (
63
+ ConfigurationError,
64
+ ContractViolation,
65
+ EntrypointError,
66
+ FlywheelError,
67
+ LeaseLost,
68
+ ModeNotAvailable,
69
+ RoutingNotInstallable,
70
+ TransportError,
71
+ )
72
+ from .event_stream import CommandOutputAdapter
73
+ from .execution_identity import ExecutionChild, ExecutionSnapshot
74
+ from .host_description import WARMUP_PROMPT, warm_up_command
75
+
76
+ # Re-exported, not used in this module. The redundant alias is the
77
+ # explicit-re-export form: it keeps `from agent_flywheel import
78
+ # read_discovered_agent` working -- which it does today -- without adding the
79
+ # name to __all__ and thereby claiming it as a supported public surface.
80
+ from .host_description import read_discovered_agent as read_discovered_agent
81
+ from .introspection import DiscoveredAgent, ToolDefinition
82
+ from .introspection import introspect as introspect_agent
83
+ from .isolation import unroutable_providers_present
84
+ from .outcomes import (
85
+ CaptureScope,
86
+ Trajectory,
87
+ current_capture,
88
+ current_rollout_id,
89
+ record_tool_call,
90
+ )
91
+ from .policy import (
92
+ ARCHIVE_REF_PREFIXES,
93
+ PolicySnapshot,
94
+ PolicySource,
95
+ is_archive_ref,
96
+ missing_credentials_cause,
97
+ policy_source,
98
+ )
99
+ from .production import Attachment, CaptureVerdict, attach, capture_verdict
100
+ from .prompt import (
101
+ SERVABLE_PROMPT_STATUSES,
102
+ PromptDecision,
103
+ prompt_decision,
104
+ )
105
+ from .runner import RolloutRunner
106
+ from .transport import AttachTransport, FlywheelTransport
107
+
108
+ __version__ = "0.1.0"
109
+
110
+ #: This package's logger. It has no handler and never installs one -- it
111
+ #: emits into whatever logging the customer's process already configured,
112
+ #: exactly as the OpenTelemetry extra emits into their TracerProvider.
113
+ #: Used for the answers a caller CANNOT see in the return value: `{}` from
114
+ #: `endpoint_kwargs` and `current_prompt` is the same value for "nothing is
115
+ #: deployed" and "you never told me where the control plane is", and only
116
+ #: one of those is a thing an operator can fix.
117
+ _log = logging.getLogger(__name__)
118
+
119
+ __all__ = [
120
+ "ARCHIVE_REF_PREFIXES",
121
+ "CONTRACT_VERSION",
122
+ "SERVABLE_PROMPT_STATUSES",
123
+ "TASK",
124
+ "AttachTransport",
125
+ "Attachment",
126
+ "CaptureScope",
127
+ "CaptureVerdict",
128
+ "CommandOutputAdapter",
129
+ "ConfigurationError",
130
+ "ContractViolation",
131
+ "DiscoveredAgent",
132
+ "EntrypointError",
133
+ "ExclusionReason",
134
+ "ExecutionChild",
135
+ "ExecutionSnapshot",
136
+ "FlywheelError",
137
+ "FlywheelTransport",
138
+ "InputShape",
139
+ "LeaseLost",
140
+ "ModeNotAvailable",
141
+ "Outcome",
142
+ "PolicySnapshot",
143
+ "PolicySource",
144
+ "PromptDecision",
145
+ "RolloutOutput",
146
+ "RolloutRequest",
147
+ "RolloutRunner",
148
+ "RoutingNotInstallable",
149
+ "ToolCallOutcome",
150
+ "ToolDefinition",
151
+ "Trajectory",
152
+ "TransportError",
153
+ "__version__",
154
+ "apply_current_policy",
155
+ "attach",
156
+ "capture_verdict",
157
+ "current_capture",
158
+ "current_mode",
159
+ "current_policy",
160
+ "current_prompt",
161
+ "current_rollout_id",
162
+ "endpoint_kwargs",
163
+ "introspect_agent",
164
+ "is_archive_ref",
165
+ "policy_source",
166
+ "prompt_decision",
167
+ "record_tool_call",
168
+ "resolve_entrypoint",
169
+ "resolve_mode",
170
+ "serve",
171
+ ]
172
+
173
+ #: Every environment variable this package reads. Two of them select
174
+ #: behaviour -- PERCEPTEYE_AGENT_MODE and PERCEPTEYE_INTROSPECT -- and the rest are
175
+ #: addresses and credentials.
176
+ #:
177
+ #: PERCEPTEYE_AGENT_MODE names what the AGENT is for -- training or serving
178
+ #: users -- rather than naming a knob on this SDK, which is why "agent" is in
179
+ #: it. It carries the prefix because an unprefixed AGENT_MODE is a name we
180
+ #: would not own: it is plausible enough that a customer already exports it for
181
+ #: something unrelated, and an unrecognised value RAISES here, so a collision
182
+ #: would stop a flywheel over a word that was never addressed to us. The prefix
183
+ #: buys the right to refuse a value we cannot parse.
184
+ #:
185
+ #: Being inside the namespace also means `build_child_env` strips it from a
186
+ #: rollout child along with everything else PERCEPTEYE_, which is correct: it
187
+ #: configures the process that DONATES capacity, and a child does not call
188
+ #: serve().
189
+ ENV_CONTROL_PLANE_URL = "PERCEPTEYE_CONTROL_PLANE_URL"
190
+ ENV_API_KEY = "PERCEPTEYE_API_KEY"
191
+ ENV_MODE = "PERCEPTEYE_AGENT_MODE"
192
+ ENV_ROLLOUT_ID = "PERCEPTEYE_ROLLOUT_ID" # set by us, for the child
193
+ ENV_TRAJECTORY_DIR = "PERCEPTEYE_TRAJECTORY_DIR" # set by us, for the child
194
+
195
+ #: What this PROCESS is for. ``"rollout"`` is the legacy spelling of
196
+ #: ``"training"`` and still means exactly that.
197
+ #:
198
+ #: This used to be ``Literal["rollout"]`` -- one value, which `serve()` checked
199
+ #: and rejected everything else, so the parameter carried zero bits and the
200
+ #: field of the same name on the registration payload carried zero bits with
201
+ #: it. It is now the deployment switch, because the alternative was a SECOND
202
+ #: thing called "mode" sitting next to a vestigial first one.
203
+ FlywheelMode = Literal["training", "production", "rollout"]
204
+
205
+ #: Accepted spellings, resolved to the two real modes. Deliberately a small
206
+ #: closed set: an unrecognised value RAISES rather than guessing, because the
207
+ #: two ways of guessing wrong are "silently contribute capacity the operator meant
208
+ #: to switch off" and "silently contribute nothing while they believe training is
209
+ #: running". Neither is recoverable by the operator, because both look exactly
210
+ #: like success.
211
+ _MODE_ALIASES: dict[str, str] = {
212
+ "training": "training",
213
+ "train": "training",
214
+ "rollout": "training", # the legacy value, unchanged in meaning
215
+ "rollouts": "training",
216
+ "fine-tuning": "training",
217
+ "finetuning": "training",
218
+ "improving": "training",
219
+ "production": "production",
220
+ "prod": "production",
221
+ "serving": "production",
222
+ }
223
+
224
+
225
+ def _mode_from(raw: Any, *, source: str) -> str:
226
+ """One spelling to one mode, or a refusal that names where it came from."""
227
+ key = str(raw).strip().lower()
228
+ if key not in _MODE_ALIASES:
229
+ raise ConfigurationError(
230
+ f"{source}{raw!r} is not a mode. Use 'training' (this process "
231
+ f"does training work: it claims rollouts and runs them) or "
232
+ f"'production' (it does not). Recognised spellings: "
233
+ f"{', '.join(sorted(_MODE_ALIASES))}."
234
+ )
235
+ return _MODE_ALIASES[key]
236
+
237
+
238
+ def _resolve_mode_source(explicit: str | None = None) -> tuple[str, str]:
239
+ """The mode, and which of ``argument``/``environment``/``default`` set it.
240
+
241
+ An EMPTY explicit argument falls through to the environment rather than
242
+ being honoured as a value. ``mode=cfg.get("percepteye_mode", "")`` is the
243
+ ordinary way a config layer says "not set", and treating it as a value
244
+ silently pinned the process to "training" without ever reading
245
+ ``PERCEPTEYE_AGENT_MODE`` -- so an operator who had flipped the switch to stop
246
+ contributing kept contributing, on their own credentials, with nothing printed.
247
+ That is the exact failure ``_MODE_ALIASES`` calls unrecoverable, arrived at
248
+ through the one input that skipped the closed-set check entirely.
249
+ """
250
+ if explicit is not None and str(explicit).strip() != "":
251
+ return _mode_from(explicit, source="mode="), "argument"
252
+ env = os.environ.get(ENV_MODE)
253
+ if env is not None and str(env).strip() != "":
254
+ return _mode_from(env, source=f"{ENV_MODE}="), "environment"
255
+ return "training", "default"
256
+
257
+
258
+ def resolve_mode(explicit: str | None = None) -> str:
259
+ """Which mode this process is in: ``"training"`` or ``"production"``.
260
+
261
+ Precedence is explicit argument, then ``PERCEPTEYE_AGENT_MODE``, then
262
+ ``"training"``. The argument wins so that a caller who has hard-coded a
263
+ mode gets what they wrote; the environment provides the default in the
264
+ ordinary case where nobody passed one, which is what lets an operator flip
265
+ a deployment WITHOUT editing code. An empty or whitespace argument counts
266
+ as "nobody passed one".
267
+
268
+ Raises:
269
+ ConfigurationError: for any value that is not a recognised mode. There
270
+ is no safe guess -- see ``_MODE_ALIASES``.
271
+ """
272
+ return _resolve_mode_source(explicit)[0]
273
+
274
+
275
+ def current_mode() -> str:
276
+ """The mode this process would serve in, from the environment alone.
277
+
278
+ ``"training"`` or ``"production"``. Read it to log which way a deployment
279
+ is configured, or to assert it in a smoke test. Raises for an unrecognised
280
+ ``PERCEPTEYE_AGENT_MODE``, exactly as :func:`serve` does, so a typo surfaces
281
+ wherever it is first read rather than only at startup.
282
+ """
283
+ return resolve_mode()
284
+
285
+
286
+ def _env_true(name: str, *, default: bool) -> bool:
287
+ """An env kill switch that only a DELIBERATE value can flip.
288
+
289
+ Unset means the default. ``0``/``false``/``no``/``off`` disable; anything
290
+ else is treated as enabled, so a typo cannot silently switch capture off
291
+ and leave the operator believing it is on.
292
+ """
293
+ raw = os.environ.get(name)
294
+ if raw is None or raw.strip() == "":
295
+ return default
296
+ return raw.strip().lower() not in ("0", "false", "no", "off")
297
+
298
+
299
+ def current_policy(
300
+ *,
301
+ agent_id: str,
302
+ control_plane_url: str | None = None,
303
+ api_key: str | None = None,
304
+ ) -> dict[str, Any]:
305
+ """Ask the control plane what model this agent should be running.
306
+
307
+ For the customer's OWN agent process — the one serving their users —
308
+ which is a different process from :func:`serve`. ``serve`` contributes
309
+ capacity for training; this is how the production agent picks up what
310
+ that training produced.
311
+
312
+ Returns ``{}`` when the control plane cannot answer, and callers should
313
+ read that as "keep your current configuration". An unreachable control
314
+ plane must never stop a customer's agent from serving its users.
315
+
316
+ What comes back is an ENDPOINT, because that is the only form an agent
317
+ can act on — an artifact URI is not callable. It is shaped for an
318
+ OpenAI-compatible client::
319
+
320
+ pol = fw.current_policy(agent_id="coding-agent")
321
+ ep = pol.get("endpoint")
322
+ if ep:
323
+ client = OpenAI(base_url=ep["base_url"],
324
+ api_key=os.environ[ep["api_key_env"]])
325
+ model = ep["model"]
326
+
327
+ ``endpoint`` is absent when the control plane has nothing deployed to
328
+ hand you — keep your current configuration, do not fall back to
329
+ something you guessed.
330
+ """
331
+ url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL) or ""
332
+ key = api_key or os.environ.get(ENV_API_KEY) or ""
333
+ cause = missing_credentials_cause(url, key)
334
+ if cause:
335
+ _log.warning(
336
+ "agent-flywheel: not reading the served policy for agent_id=%r: %s, so "
337
+ "the control plane was never asked. This agent keeps the model it "
338
+ "is on.", agent_id, cause,
339
+ )
340
+ return {}
341
+ from ._http import ControlPlaneClient
342
+
343
+ client = ControlPlaneClient(base_url=url, api_key=key)
344
+ try:
345
+ from .transport import AttachTransport
346
+
347
+ return AttachTransport(client, agent_id=agent_id).policy_current()
348
+ finally:
349
+ client.close()
350
+
351
+
352
+ def current_prompt(
353
+ *,
354
+ agent_id: str,
355
+ control_plane_url: str | None = None,
356
+ api_key: str | None = None,
357
+ ) -> dict[str, Any]:
358
+ """The approved system prompt this agent should be running, or ``{}``.
359
+
360
+ The other half of :func:`current_policy`. That one answers "which model";
361
+ training also produces a system PROMPT, co-optimized with that model and
362
+ gated by the same human approval — and until this existed an agent could
363
+ learn only that the prompt had moved, never what it moved to. The snapshot
364
+ in :class:`PolicySource` says as much in its own comment: its generation
365
+ changes for a prompt change, "which this snapshot does not carry and a
366
+ client rebuild would not apply".
367
+
368
+ Returns the deliverable::
369
+
370
+ {"name": ..., "version": 3, "sha256": ..., "text": "...",
371
+ "candidate_kind": "replacement", "status": "approved",
372
+ "model_prompt_checksum": ..., "stale_baseline": false}
373
+
374
+ ``{}`` means KEEP THE PROMPT YOU HAVE — nothing has passed the approval
375
+ gate for this agent, or the control plane could not answer. It never means
376
+ "use no prompt", and an unreachable control plane must never blank a
377
+ running agent's instructions.
378
+
379
+ **Deliberately not applied for you, and that is the difference from**
380
+ :func:`apply_current_policy`. Swapping a model endpoint is invisible to an
381
+ agent's behaviour; swapping its system prompt changes what it DOES. That is
382
+ a decision for the person who owns the agent, so this hands back the text
383
+ and stops.
384
+
385
+ **Do not re-derive whether it may be applied — ask** :func:`prompt_decision`.
386
+ This docstring used to list the fields to check and the order to check them
387
+ in, which made every caller reimplement a rule with no mechanism behind it.
388
+ The rule is a function now::
389
+
390
+ served = fw.current_prompt(agent_id="my-agent")
391
+ decision = fw.prompt_decision(served)
392
+ if decision.apply:
393
+ system_prompt = decision.text # never None here
394
+ else:
395
+ log.info("keeping the current prompt: %s", decision.reason)
396
+
397
+ It answers the same four questions the Node SDK's ``promptDecision`` asks,
398
+ in the same order and with the same sentences, so the two packages refuse
399
+ the same bundles for the same stated reason. See
400
+ :mod:`agent_flywheel.prompt`.
401
+ """
402
+ url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL) or ""
403
+ key = api_key or os.environ.get(ENV_API_KEY) or ""
404
+ cause = missing_credentials_cause(url, key)
405
+ if cause:
406
+ # SAID, because `prompt_decision({})` blames the control plane for
407
+ # serving no approved prompt -- true of the empty answer and wrong
408
+ # about why, which sends an operator to the dashboard to look for an
409
+ # approval that was never the problem.
410
+ _log.warning(
411
+ "agent-flywheel: not reading the served prompt for agent_id=%r: %s, so "
412
+ "the control plane was never asked. This agent keeps the prompt it "
413
+ "assembles itself.", agent_id, cause,
414
+ )
415
+ return {}
416
+ from ._http import ControlPlaneClient
417
+
418
+ client = ControlPlaneClient(base_url=url, api_key=key)
419
+ try:
420
+ from .transport import AttachTransport
421
+
422
+ tr = AttachTransport(client, agent_id=agent_id)
423
+ ask = getattr(tr, "prompt_current", None)
424
+ return (ask() if callable(ask) else {}) or {}
425
+ finally:
426
+ client.close()
427
+
428
+
429
+ def endpoint_kwargs(
430
+ policy: Mapping[str, Any] | None, *, api_key: str | None = None,
431
+ ) -> dict[str, Any]:
432
+ """The ONE place a ``/policy/current`` body becomes client kwargs.
433
+
434
+ Extracted because there are now two callers — :func:`apply_current_policy`
435
+ and :class:`PolicySource` — and "a caller who unpacks it slightly
436
+ differently is a caller sampling from something other than what was
437
+ trained" applies at least as much between two of OUR surfaces as it does
438
+ between two of the customer's.
439
+
440
+ ``{}`` means nothing to apply: no endpoint, no ``base_url``, or no
441
+ credential we can authenticate with. That last one is deliberate — an
442
+ endpoint we cannot authenticate to is not usable, and returning it would
443
+ produce 401s at the caller's first request rather than a legible "nothing
444
+ to apply" here.
445
+ """
446
+ if not isinstance(policy, Mapping):
447
+ return {}
448
+ ep = policy.get("endpoint")
449
+ if not isinstance(ep, Mapping) or not ep.get("base_url"):
450
+ return {}
451
+ # PRECEDENCE, and the order is the contract:
452
+ #
453
+ # 1. an explicit ``api_key=`` argument — the caller's own override always
454
+ # wins, so a customer who wants full control keeps it;
455
+ # 2. ``endpoint.api_key`` from the control plane, when it sent one;
456
+ # 3. the environment variable the control plane NAMED;
457
+ # 4. ``PERCEPTEYE_API_KEY``.
458
+ #
459
+ # (2) exists for one case, and without it that case fails silently. A
460
+ # checkpoint promotion normally moves only ``model`` — base_url and the
461
+ # credential stay put, so an agent upgrades with nothing new. They do NOT
462
+ # stay put when the control plane moves an agent to a DIFFERENT provider:
463
+ # then base_url and the credential both change, and an agent cannot read an
464
+ # env var it was never given. Without (2) this function returns ``{}``, the
465
+ # caller keeps its previous configuration, and the agent FREEZES on its
466
+ # last policy — receiving no further upgrades and reporting nothing wrong.
467
+ #
468
+ # The value is never persisted or logged here; it goes straight into the
469
+ # client kwargs the caller builds, exactly like a key read from the
470
+ # environment.
471
+ key_env = str(ep.get("api_key_env") or ENV_API_KEY)
472
+ key = (
473
+ api_key
474
+ or str(ep.get("api_key") or "").strip()
475
+ or os.environ.get(key_env)
476
+ or os.environ.get(ENV_API_KEY)
477
+ )
478
+ if not key:
479
+ return {}
480
+ model = str(ep.get("model") or "")
481
+ # AN ARCHIVE URI IS NOT A MODEL NAME, and this is the model half of the
482
+ # pair rule `prompt_decision` enforces for the prompt half. The control
483
+ # plane already refuses to serve one; re-asserted here on the same argument
484
+ # as `SERVABLE_PROMPT_STATUSES`, because a candidate whose checkpoint has
485
+ # been uploaded but not yet deployed behind a serving endpoint would
486
+ # otherwise be handed to `OpenAI(model="gs://...")` and 404 on every
487
+ # completion. Refusing keeps the champion, which is what every other gate
488
+ # on this path fails closed to.
489
+ #
490
+ # SAID OUT LOUD, because silence here is indistinguishable from "nothing is
491
+ # deployed": `{}` is this function's ordinary answer and the caller is
492
+ # documented to read it as "keep your current configuration". An operator
493
+ # whose promotion never arrives needs the sentence, not the empty dict.
494
+ if is_archive_ref(model):
495
+ _log.warning(
496
+ "agent-flywheel: NOT applying the served policy for this agent: the "
497
+ "served model %r is a stored-artifact reference, not a name an "
498
+ "endpoint can serve; that address would 404 on every completion. "
499
+ "Keeping the model you are on.", model,
500
+ )
501
+ return {}
502
+ return {
503
+ "base_url": str(ep["base_url"]),
504
+ "api_key": str(key),
505
+ "model": model,
506
+ }
507
+
508
+
509
+ def apply_current_policy(
510
+ *,
511
+ agent_id: str,
512
+ control_plane_url: str | None = None,
513
+ api_key: str | None = None,
514
+ set_env: bool = False,
515
+ ) -> dict[str, Any]:
516
+ """Turn :func:`current_policy` into arguments you can hand a client.
517
+
518
+ The last mile. ``current_policy`` already returns the right answer, but
519
+ every caller had to unpack it the same way, and a caller who unpacks it
520
+ slightly differently is a caller sampling from something other than what
521
+ was trained. Returns kwargs shaped for an OpenAI-compatible client::
522
+
523
+ kw = fw.apply_current_policy(agent_id="coding-agent")
524
+ if kw:
525
+ client, model = OpenAI(**{k: v for k, v in kw.items()
526
+ if k != "model"}), kw["model"]
527
+
528
+ Returns ``{}`` when nothing is deployed or the control plane cannot
529
+ answer — keep your current configuration. Same fail-open as
530
+ ``current_policy``, for the same reason: an unreachable control plane must
531
+ never stop a customer's agent from serving its users.
532
+
533
+ NOT FOR :func:`serve`, and that is a correctness boundary rather than a
534
+ style preference. A ``serve`` rollout samples through the per-rollout
535
+ GATEWAY url, which is what attributes the completion to that rollout and
536
+ captures it — and the gateway already rewrites the model field to the
537
+ policy under test. Pointing a donor agent at the policy endpoint directly
538
+ would bypass capture entirely and silently produce a run with nothing to
539
+ train on. Use this in your PRODUCTION agent, the one serving your users.
540
+
541
+ ``set_env=True`` additionally exports ``OPENAI_BASE_URL`` /
542
+ ``OPENAI_API_KEY`` for libraries that read the environment rather than
543
+ accept arguments. Off by default: mutating the process environment is a
544
+ side effect a caller should ask for.
545
+ """
546
+ pol = current_policy(
547
+ agent_id=agent_id, control_plane_url=control_plane_url, api_key=api_key,
548
+ )
549
+ out = endpoint_kwargs(pol, api_key=api_key)
550
+ if not out:
551
+ return {}
552
+ if set_env:
553
+ os.environ["OPENAI_BASE_URL"] = out["base_url"]
554
+ os.environ["OPENAI_API_KEY"] = out["api_key"]
555
+ return out
556
+
557
+
558
+ def _describe_via_host(
559
+ command: list[Any], *, artifacts_dir: str, agent_id: str,
560
+ execution_snapshot: ExecutionSnapshot | None = None,
561
+ endpoint_env: tuple[str, str] | None = None,
562
+ ) -> DiscoveredAgent | None:
563
+ """Have the host describe a command agent, by running it once.
564
+
565
+ Returns None when the host produced nothing, having said why. The failure
566
+ is silent everywhere else -- the plugin's hook simply never fires -- and an
567
+ operator who is not told here finds out as an empty workflow list an hour
568
+ later, with nothing in between naming the cause.
569
+ """
570
+ # The TASK sentinel marks where a rollout's task goes in the argv. There is
571
+ # no rollout here, so it is filled with the warm-up prompt: an argv still
572
+ # holding the sentinel would be passed to the binary as the literal string.
573
+ argv = [WARMUP_PROMPT if a is TASK else str(a) for a in command]
574
+ warm_dir = os.path.join(artifacts_dir, "_describe")
575
+
576
+ snapshot = execution_snapshot
577
+ snapshot_child: ExecutionChild | None = None
578
+ try:
579
+ env = None
580
+ cwd = None
581
+ if snapshot is not None:
582
+ # Registration must describe the same scaffold the rollouts use.
583
+ # Letting this one preflight read ambient HOME would register a
584
+ # prompt/tool identity that no trainable child actually ran.
585
+ try:
586
+ snapshot_child = snapshot.prepare_child("describe")
587
+ except Exception as exc:
588
+ raise ConfigurationError(
589
+ "execution snapshot could not prepare the discovery child: "
590
+ f"{type(exc).__name__}"
591
+ ) from exc
592
+ env = dict(os.environ)
593
+ cwd = warm_dir
594
+ try:
595
+ rollout_environment = dict(snapshot_child.environment)
596
+ discovery_environment = dict(
597
+ snapshot_child.discovery_environment
598
+ )
599
+ except Exception as exc:
600
+ raise ConfigurationError(
601
+ f"{snapshot.adapter_name} execution child environments "
602
+ "must be string mappings"
603
+ ) from exc
604
+ if any(
605
+ not isinstance(name, str) or not isinstance(value, str)
606
+ for mapping in (rollout_environment, discovery_environment)
607
+ for name, value in mapping.items()
608
+ ):
609
+ raise ConfigurationError(
610
+ f"{snapshot.adapter_name} execution child environments "
611
+ "must contain only string names and values"
612
+ )
613
+ if any(
614
+ rollout_environment.get(name) != value
615
+ for name, value in discovery_environment.items()
616
+ ):
617
+ raise ConfigurationError(
618
+ f"{snapshot.adapter_name} discovery environment must be a "
619
+ "non-secret subset of its rollout environment"
620
+ )
621
+ else:
622
+ discovery_environment = None
623
+ described = warm_up_command(
624
+ argv, trajectory_dir=warm_dir, agent_name=agent_id, env=env, cwd=cwd,
625
+ endpoint_env=endpoint_env,
626
+ isolation_environment=(
627
+ discovery_environment
628
+ ),
629
+ )
630
+ finally:
631
+ if snapshot_child is not None:
632
+ lifecycle_failure: Exception | None = None
633
+ try:
634
+ assert snapshot is not None
635
+ snapshot.attest_child(snapshot_child)
636
+ except Exception as exc:
637
+ lifecycle_failure = exc
638
+ try:
639
+ snapshot.release_child(snapshot_child)
640
+ except Exception as exc:
641
+ lifecycle_failure = lifecycle_failure or exc
642
+ if lifecycle_failure is not None:
643
+ raise ConfigurationError(
644
+ "execution snapshot lifecycle failed during command-agent "
645
+ f"discovery ({type(lifecycle_failure).__name__}); "
646
+ "registration was refused"
647
+ ) from lifecycle_failure
648
+ if described is not None:
649
+ n = described.tool_definitions
650
+ print(
651
+ f"[percepteye] described {agent_id} from the host: "
652
+ f"{len(n) if n is not None else 'no'} tool(s), "
653
+ f"{len(described.system_prompt or '')} chars of system prompt"
654
+ # WHICH SOURCE answered, named rather than inferred. The two carry
655
+ # different amounts -- the wire drops read-only annotations and
656
+ # sub-agents -- so an operator reading a thin catalogue needs to
657
+ # know whether that is the agent or the vantage point.
658
+ f" (via {'the wire' if described.source_type == 'host_wire' else 'the host plugin'})."
659
+ )
660
+ return described
661
+
662
+ warnings.warn(
663
+ f"Could not describe {agent_id!r}: the agent ran, its host wrote no "
664
+ f"description, and no completion request it made declared any tools. "
665
+ f"This agent will register as introspection_state=unreadable, no "
666
+ f"workflows will be generated for it, and every phase downstream of "
667
+ f"that (reward suites, graded rollouts, training) has nothing to work "
668
+ f"from.\n"
669
+ f"Two things to check, in this order:\n"
670
+ f" 1. Does the agent reach its provider from an env var? The warm-up "
671
+ f"points OPENAI_BASE_URL / OPENAI_API_BASE / ANTHROPIC_BASE_URL at a "
672
+ f"local endpoint and reads the tool catalogue off the request. An "
673
+ f"agent that resolves its endpoint some other way never arrives, and "
674
+ f"is also not routable for rollouts -- pass endpoint_env= to serve().\n"
675
+ f" 2. Does it declare any tools at all? A CLI agent whose extensions "
676
+ f"are configured per-profile can start with an empty catalogue; goose, "
677
+ f"for one, sends zero tools unless an extension is enabled.\n"
678
+ f"For OpenClaw specifically, the usual cause is the conversation-hook "
679
+ f"opt-in: the host silently refuses to register `llm_input` for a "
680
+ f"non-bundled plugin unless the config sets\n"
681
+ f" plugins.entries.agent-flywheel.hooks."
682
+ f"allowConversationAccess = true",
683
+ RuntimeWarning,
684
+ stacklevel=3,
685
+ )
686
+ return None
687
+
688
+
689
+ def serve(
690
+ entrypoint: Any,
691
+ *,
692
+ agent_id: str,
693
+ input_shape: InputShape,
694
+ mode: FlywheelMode | None = None,
695
+ transport: Literal["attach"] = "attach",
696
+ control_plane_url: str | None = None,
697
+ api_key: str | None = None,
698
+ artifacts_dir: str | None = None,
699
+ isolation: Literal["subprocess", "inprocess"] = "subprocess",
700
+ concurrency: int = 1,
701
+ rollout_timeout_s: float = 900.0,
702
+ #: How this agent's CLI spells its completion-gate option, e.g.
703
+ #: ``"--autonomous-gate"``. Required before any gate is passed: the flag is
704
+ #: a fact about the binary and cannot be guessed, and handing an unknown
705
+ #: option to an arbitrary agent would break it. The gate COMMAND itself is
706
+ #: a property of the task and arrives per rollout. Leave unset for an agent
707
+ #: with no gate concept — it then behaves exactly as before.
708
+ gate_flag: str | None = None,
709
+ #: The environment variables THIS agent reads its endpoint from, when they
710
+ #: are not the OpenAI/Anthropic names set by default. Declared, never
711
+ #: sniffed: an agent that reads neither default looks identical from here
712
+ #: to one that is correctly routed, right up until the completions arrive
713
+ #: at the live provider instead of the gateway.
714
+ endpoint_env: tuple[str, str] | None = None,
715
+ command_output_adapter: CommandOutputAdapter | None = None,
716
+ execution_snapshot: ExecutionSnapshot | None = None,
717
+ harness_snapshot: ExecutionSnapshot | None = None,
718
+ attested_execution_components: Mapping[str, str] | None = None,
719
+ reports_tool_calls: bool = False,
720
+ method: str | None = None,
721
+ adapters: list[str] | str | None = "auto",
722
+ introspect: bool = True,
723
+ max_rollouts: int | None = None,
724
+ on_event: Callable[[str, dict[str, Any]], None] | None = None,
725
+ ) -> int:
726
+ """Claim rollouts and run them until stopped. Returns rollouts completed.
727
+
728
+ Everything that can be wrong with the configuration is checked here,
729
+ before the first rollout: the entrypoint resolves, the credential exists,
730
+ the control plane answers, and the isolation mode is compatible with the
731
+ entrypoint you passed.
732
+
733
+ Args:
734
+ entrypoint: ``"module:attr"`` (required for subprocess isolation, so a
735
+ child can rebuild the agent) or a constructed object
736
+ (``isolation="inprocess"`` only).
737
+ mode: ``"training"`` to contribute capacity, ``"production"`` to do
738
+ nothing. Omit it -- which is the point -- and ``PERCEPTEYE_AGENT_MODE``
739
+ decides, so the same deployed image serves or does not without an
740
+ edit. Unset means ``"training"``, preserving the behaviour of every
741
+ existing call. ``"rollout"`` is the legacy spelling of
742
+ ``"training"``. In production mode this returns 0 before opening
743
+ any connection, and prints one line saying so; it does NOT raise,
744
+ because a switch that crashes the service it was meant to quiet is
745
+ not a switch. An unrecognised value raises
746
+ :class:`ConfigurationError` rather than guessing.
747
+ input_shape: how ``turn_input`` reaches your agent -- ``"text"``,
748
+ ``"messages"`` or ``"dict"``. Declared, never guessed.
749
+ adapters: framework integrations, by NAME, bound inside whichever
750
+ process runs your agent. ``"auto"`` (the default) binds every
751
+ built-in whose framework is installed, so a LangChain or LangGraph
752
+ agent reports per-call outcomes without you writing anything.
753
+ Pass ``[]`` to bind none, or ``["langchain"]`` to require one --
754
+ a NAMED adapter that cannot bind raises, because naming it asserts
755
+ that it is how your calls are captured.
756
+ reports_tool_calls: assert that this flywheel reports a per-call
757
+ outcome for every tool call it makes. Without it, the reward
758
+ surface available to your rollouts is materially smaller, because
759
+ the verifiers that grade "did this call land" have nothing to
760
+ read and correctly abstain.
761
+ execution_snapshot: framework-neutral execution-adapter lifecycle.
762
+ The adapter supplies opaque exact component digests, prepares an
763
+ isolated child view, and attests/releases it after every rollout.
764
+ Command and import-spec children receive its environment mapping;
765
+ an in-process object requires an empty mapping because process-wide
766
+ environment mutation is not isolated. An incomplete adapter never
767
+ yields an OPSD-eligible execution identity.
768
+ harness_snapshot: deprecated compatibility alias for
769
+ ``execution_snapshot``. Prime-specific construction lives in the
770
+ optional ``adapters.prime_agent`` module; core treats it like any
771
+ other execution adapter.
772
+ command_output_adapter: optional command-specific stdout projection.
773
+ Core never guesses a host vocabulary. An adapter may project final
774
+ text, independently observed model-call count, tool outcomes, and
775
+ explicit exclusion evidence.
776
+ attested_execution_components: framework-neutral opaque evidence from
777
+ an execution adapter, as stable names mapped to lowercase 64-hex
778
+ SHA-256 values. The SDK interprets none of the names: it validates,
779
+ sorts, and folds them into ``agent_fingerprint.execution_sha256``
780
+ with the discovered prompt/tools/sub-agents/model and serve config.
781
+ An adapter supplying a digest is asserting it covers mutable state
782
+ discovery cannot observe; do not put secrets or free-form values in
783
+ this mapping. Without at least one adapter-owned component (or a
784
+ component-providing ``execution_snapshot``/``harness_snapshot``),
785
+ the SDK withholds the
786
+ OPSD digest rather than claiming discovery identifies unobserved
787
+ implementation code.
788
+ max_rollouts: stop after this many rollouts COMPLETE, and return. The
789
+ only clean exit this function has: nothing in this package handles
790
+ SIGINT or SIGTERM, so otherwise it runs until the process is
791
+ killed and any in-flight rollout is orphaned rather than abandoned,
792
+ to be reclaimed by lease expiry. Note the count is completions, not
793
+ successes -- a crash and a graded pass both increment it.
794
+ on_event: ``Callable[[str, dict], None]``, called SYNCHRONOUSLY on
795
+ whichever thread produced the event -- the serving thread for
796
+ ``claim_*``, a pool worker for the per-rollout kinds, and the
797
+ heartbeat thread for ``lease_lost``/``heartbeat_failed`` -- so a
798
+ slow callback slows that thread. Exceptions raised by the callback
799
+ are SWALLOWED: it is an observability hook and must not be able to
800
+ change what it observes. Keep it cheap and non-blocking; the
801
+ heartbeat thread renews every in-flight lease. Kinds:
802
+ ``claim_empty``, ``claim_failed``,
803
+ ``rollout_start``, ``rollout_reported``, ``rollout_crash``,
804
+ ``report_failed``, ``artifact_failed``, ``lease_lost``,
805
+ ``heartbeat_failed``. Without it a flywheel is silent, and an empty
806
+ queue is indistinguishable from a broken claim route --
807
+ ``claim_empty`` versus ``claim_failed`` is what separates them.
808
+ """
809
+ # ── the deployment switch ────────────────────────────────────────────────
810
+ # Resolved FIRST, before the credential checks and before any socket, so
811
+ # that a production deployment needs neither a control-plane URL nor a key
812
+ # to start clean. Turning training off must not require being configured
813
+ # for it.
814
+ resolved_mode, _mode_source = _resolve_mode_source(mode)
815
+
816
+ # An override has to be announced in BOTH directions. The production
817
+ # branch below prints when code says stop; this prints when code says GO
818
+ # while the environment said stop -- which is the direction an operator
819
+ # actually cares about, and the one that was silent. Flipping the variable
820
+ # and seeing the flywheel keep claiming, with nothing on stdout to explain
821
+ # it, is the same silence this function refuses everywhere else.
822
+ if _mode_source == "argument":
823
+ _env_raw = os.environ.get(ENV_MODE)
824
+ if _env_raw is not None and str(_env_raw).strip() != "":
825
+ try:
826
+ _env_mode = _mode_from(_env_raw, source=f"{ENV_MODE}=")
827
+ except ConfigurationError as exc:
828
+ # Not fatal: the argument is authoritative and this process has
829
+ # been told what to do. But an unreadable switch must not pass
830
+ # unremarked, or the next operator to rely on it is misled.
831
+ warnings.warn(str(exc), RuntimeWarning, stacklevel=2)
832
+ _env_mode = None
833
+ if _env_mode is not None and _env_mode != resolved_mode:
834
+ print(
835
+ f"[percepteye] mode={resolved_mode} — set by mode= in "
836
+ f"code, overriding {ENV_MODE}={_env_mode}. Remove mode= "
837
+ f"from the call to let this deployment decide.",
838
+ flush=True,
839
+ )
840
+
841
+ if resolved_mode == "production":
842
+ # Return, never raise: this is a running service, and a switch that
843
+ # crashes the process it was meant to quiet is not a switch. Return 0
844
+ # because zero rollouts is the literal truth.
845
+ #
846
+ # And never SILENTLY: a no-op serve() that printed nothing is the exact
847
+ # failure this package refuses everywhere else -- the operator believes
848
+ # capacity is being contributed and nothing says otherwise. One line, on
849
+ # the way out, naming what decided it.
850
+ print(
851
+ f"[percepteye] mode=production — not claiming rollouts. "
852
+ f"(set {ENV_MODE}=training to contribute capacity"
853
+ + (", or remove mode= from this call"
854
+ if _mode_source == "argument" else "")
855
+ + ")",
856
+ flush=True,
857
+ )
858
+ return 0
859
+ if transport != "attach":
860
+ raise ModeNotAvailable(
861
+ f"transport={transport!r} is not available in contract {CONTRACT_VERSION}."
862
+ )
863
+
864
+ url = control_plane_url or os.environ.get(ENV_CONTROL_PLANE_URL)
865
+ if not url:
866
+ raise ConfigurationError(
867
+ f"control_plane_url is required (or set {ENV_CONTROL_PLANE_URL})"
868
+ )
869
+ key = api_key or os.environ.get(ENV_API_KEY)
870
+ if not key:
871
+ raise ConfigurationError(f"an API key is required (or set {ENV_API_KEY})")
872
+
873
+ adir = artifacts_dir or os.path.join(os.getcwd(), ".percepteye-artifacts")
874
+ os.makedirs(adir, exist_ok=True)
875
+
876
+ if execution_snapshot is not None and harness_snapshot is not None:
877
+ raise ConfigurationError(
878
+ "pass execution_snapshot or the deprecated harness_snapshot alias, not both"
879
+ )
880
+ if harness_snapshot is not None:
881
+ warnings.warn(
882
+ "serve(harness_snapshot=...) is deprecated; use execution_snapshot=...",
883
+ DeprecationWarning,
884
+ stacklevel=2,
885
+ )
886
+ resolved_execution_snapshot = (
887
+ execution_snapshot if execution_snapshot is not None else harness_snapshot
888
+ )
889
+
890
+ from ._http import ControlPlaneClient
891
+
892
+ client = ControlPlaneClient(url, key)
893
+ tr = AttachTransport(client, agent_id)
894
+
895
+ # Fail at startup, loudly. A health check nobody calls is not a health
896
+ # check; an earlier SDK shipped one with zero call sites.
897
+ health = tr.health()
898
+ served = health.get("contract_versions_supported") or [health.get("contract_version")]
899
+ if CONTRACT_VERSION not in [v for v in served if v]:
900
+ raise ConfigurationError(
901
+ f"control plane speaks {served}, this SDK speaks {CONTRACT_VERSION}. "
902
+ f"Upgrade agent-flywheel."
903
+ )
904
+
905
+ # Provider configuration we cannot point at the gateway. Reported here, in
906
+ # front of the person who set it, rather than a run later as a
907
+ # training-grade exclusion nobody can explain. A warning and not a refusal:
908
+ # a name on this list is evidence, not proof -- the agent may not use that
909
+ # provider at all -- and the control plane's reported-vs-served call count
910
+ # is the check that actually decides.
911
+ stripped = unroutable_providers_present()
912
+ if stripped:
913
+ warnings.warn(
914
+ f"{', '.join(stripped)} set in this environment. The PerceptEye "
915
+ f"gateway speaks the OpenAI completions API, so these cannot be "
916
+ f"routed through it, and they are REMOVED from each rollout's "
917
+ f"child process. An agent built on one of those providers will "
918
+ f"raise rather than sample from the live provider unobserved. "
919
+ f"Use an OpenAI-compatible client to have your completions "
920
+ f"captured.",
921
+ RuntimeWarning,
922
+ stacklevel=2,
923
+ )
924
+
925
+ runner = RolloutRunner(
926
+ entrypoint,
927
+ transport=tr,
928
+ input_shape=input_shape,
929
+ artifacts_dir=adir,
930
+ isolation=isolation,
931
+ concurrency=concurrency,
932
+ rollout_timeout_s=rollout_timeout_s,
933
+ gate_flag=gate_flag,
934
+ endpoint_env=endpoint_env,
935
+ command_output_adapter=command_output_adapter,
936
+ execution_snapshot=resolved_execution_snapshot,
937
+ method=method,
938
+ adapters=adapters,
939
+ on_event=on_event,
940
+ reports_tool_calls=reports_tool_calls,
941
+ attested_execution_components=attested_execution_components,
942
+ )
943
+
944
+ # A command entrypoint has no resolved object -- ``resolved`` is None by
945
+ # construction (it is already its own process). Reading ``.name`` off it
946
+ # raised AttributeError here, so every non-Python agent failed at
947
+ # registration, before its first rollout. No test caught it because the
948
+ # tests build a RolloutRunner directly and never go through serve().
949
+ entrypoint_label = (
950
+ runner.resolved.name if runner.resolved is not None
951
+ # str() each part: a TASK sentinel is not a string, and joining it
952
+ # raised TypeError here — the same class of bug the line above records
953
+ # having shipped once, where reading .name off a command entrypoint
954
+ # failed every non-Python agent at registration.
955
+ else " ".join(str(a) for a in (getattr(runner, "_command", None) or []))
956
+ or "command"
957
+ )
958
+
959
+ payload: dict[str, Any] = {
960
+ "agent_id": agent_id,
961
+ # The WIRE value, which is not the deployment mode and never was.
962
+ # `attach()` sends "production" on this same field; a donor sends
963
+ # "rollout", which names the protocol it speaks and is what intake has
964
+ # always received here. Reaching this line at all means
965
+ # resolved_mode == "training".
966
+ "mode": "rollout",
967
+ "sdk": f"agent-flywheel-python/{__version__}",
968
+ "contract_version": CONTRACT_VERSION,
969
+ "reports_tool_calls": bool(reports_tool_calls),
970
+ "isolation": isolation,
971
+ "concurrency": concurrency,
972
+ "input_shape": input_shape,
973
+ "entrypoint": entrypoint_label,
974
+ }
975
+
976
+ # ── agent discovery (ADP), from inside the process ──────────────────────
977
+ # ON by default, with two ways to refuse: ``introspect=False`` here, or
978
+ # ``PERCEPTEYE_INTROSPECT=0`` in the environment, for an operator who
979
+ # cannot edit the call site. Both are documented prominently in the
980
+ # README, because this ships your system prompt and tool catalogue to the
981
+ # control plane and a default that surprises someone is worse than no
982
+ # default at all.
983
+ #
984
+ # ``discovered_agent: None`` is sent DELIBERATELY when we could not look.
985
+ # Omitting the key would make "not introspected" indistinguishable from
986
+ # "introspected, found nothing" -- the same absence-as-zero collapse the
987
+ # tool-outcome contract exists to prevent, one layer up.
988
+ discovery_enabled = introspect and _env_true("PERCEPTEYE_INTROSPECT", default=True)
989
+ described = None
990
+ if discovery_enabled:
991
+ described = introspect_agent(
992
+ runner.resolved.fn if runner.resolved is not None else None
993
+ )
994
+ # A COMMAND agent has no object to reflect on, so the line above always
995
+ # returns None for one and the agent registers `unreadable` forever --
996
+ # which blocks workflow generation, and with it every downstream phase.
997
+ # Ask the HOST instead: run the agent once, let its own plugin observe
998
+ # the assembled prompt and the tool list it sends to the model, and
999
+ # read that. See host_description.py for why this cannot wait for a
1000
+ # rollout (the launch gate and the first rollout are mutually
1001
+ # blocking).
1002
+ if described is None and getattr(runner, "_command", None):
1003
+ described = _describe_via_host(
1004
+ runner._command, artifacts_dir=adir, agent_id=agent_id,
1005
+ execution_snapshot=runner.execution_snapshot,
1006
+ endpoint_env=endpoint_env,
1007
+ )
1008
+ payload["discovered_agent"] = described.to_wire() if described else None
1009
+
1010
+ # The rollout fingerprint is authored here, after discovery, from the same
1011
+ # framework-neutral facts registration sends plus exact pre-projection
1012
+ # discovery evidence and any adapter-attested opaque components. A disabled,
1013
+ # unreadable, or lossy description deliberately yields no digest, so an OPSD
1014
+ # consumer cannot mistake an unknown scaffold for an exact identity.
1015
+ runner.configure_execution_identity(
1016
+ described,
1017
+ discovery_enabled=discovery_enabled,
1018
+ sdk=payload["sdk"],
1019
+ contract_version=CONTRACT_VERSION,
1020
+ )
1021
+
1022
+ # Neutral worker evidence, not agent logic and not a model identity. The
1023
+ # control plane uses this authenticated registration fact to select the
1024
+ # exact rollout worker whose end-to-end latency was qualified. It still
1025
+ # verifies every returned trajectory carries the same digest; registration
1026
+ # is the pre-dispatch identity, not a substitute for observed evidence.
1027
+ if runner.execution_sha256 is not None:
1028
+ payload["agent_execution_sha256"] = runner.execution_sha256
1029
+
1030
+ tr.register(payload)
1031
+
1032
+ # ── the last mile ───────────────────────────────────────────────────────
1033
+ # Training produces a better model; this is where the agent finds out.
1034
+ # The registration response used to be discarded and nothing ever asked
1035
+ # what to run, so a trained adapter reached the customer's process only
1036
+ # by a human editing their config.
1037
+ #
1038
+ # Resolved ONCE at startup rather than polled: a policy that changes
1039
+ # under a running agent is the same confound the control plane refuses on
1040
+ # its own training runs. A change lands on the next restart, and the
1041
+ # dashboard says so.
1042
+ # getattr, not a direct call: ``policy_current`` is not on the
1043
+ # FlywheelTransport Protocol, so a third-party or older transport need
1044
+ # not implement it. The last mile must not break an agent whose
1045
+ # transport predates it — the same fail-open the method itself keeps.
1046
+ _ask = getattr(tr, "policy_current", None)
1047
+ policy = _ask() if callable(_ask) else {}
1048
+ if policy:
1049
+ # INFORMATIONAL ONLY, deliberately. A serve rollout samples through the
1050
+ # per-rollout GATEWAY url — that is what attributes and captures the
1051
+ # completion, and the gateway already rewrites the model field to the
1052
+ # policy under test. Applying the endpoint here would bypass capture
1053
+ # and produce a run with nothing to train on. Production agents call
1054
+ # apply_current_policy() instead.
1055
+ _ep = policy.get("endpoint") if isinstance(policy.get("endpoint"), dict) else {}
1056
+ print(
1057
+ f"[percepteye] policy: {policy.get('arm', 'base')} — "
1058
+ f"{policy.get('reason', '')}"
1059
+ + (f" → {_ep.get('model')} at {_ep.get('base_url')}"
1060
+ if _ep.get("base_url") else ""),
1061
+ flush=True,
1062
+ )
1063
+
1064
+ try:
1065
+ return runner.serve_forever(max_rollouts=max_rollouts)
1066
+ finally:
1067
+ client.close()