nomad-harness 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nomad/__init__.py ADDED
@@ -0,0 +1,57 @@
1
+ """Nomad: an action runtime for frontier models in physical environments.
2
+
3
+ Importing this package must stay lightweight: it never imports model-provider
4
+ SDKs, simulators, or the adapter subpackages (``nomad.models``,
5
+ ``nomad.embodiments``). ``tests/test_import_hygiene.py`` enforces this.
6
+ """
7
+
8
+ from nomad.agent import Agent
9
+ from nomad.core import (
10
+ Action,
11
+ ActionResult,
12
+ ActionStatus,
13
+ AgentContext,
14
+ Embodiment,
15
+ EmbodimentManifest,
16
+ FeedbackCode,
17
+ ModelProvider,
18
+ ModelResponse,
19
+ Observation,
20
+ ObservationMode,
21
+ OutcomeStatus,
22
+ Proposal,
23
+ RunConfig,
24
+ RunMetrics,
25
+ RunResult,
26
+ RunStatus,
27
+ StopReason,
28
+ TaskOutcome,
29
+ )
30
+ from nomad.runtime import Runtime
31
+
32
+ __version__ = "0.1.0.dev0"
33
+
34
+ __all__ = [
35
+ "Action",
36
+ "ActionResult",
37
+ "ActionStatus",
38
+ "Agent",
39
+ "AgentContext",
40
+ "Embodiment",
41
+ "EmbodimentManifest",
42
+ "FeedbackCode",
43
+ "ModelProvider",
44
+ "ModelResponse",
45
+ "Observation",
46
+ "ObservationMode",
47
+ "OutcomeStatus",
48
+ "Proposal",
49
+ "RunConfig",
50
+ "RunMetrics",
51
+ "RunResult",
52
+ "RunStatus",
53
+ "Runtime",
54
+ "StopReason",
55
+ "TaskOutcome",
56
+ "__version__",
57
+ ]
nomad/agent.py ADDED
@@ -0,0 +1,32 @@
1
+ """User-facing entry point: bind a model, an embodiment, and a run configuration."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+
7
+ from nomad.context import DEFAULT_INSTRUCTIONS
8
+ from nomad.core.protocols import Embodiment, ModelProvider
9
+ from nomad.core.run import RunConfig, RunResult
10
+ from nomad.runtime import Runtime
11
+
12
+
13
+ class Agent:
14
+ """Runs goals on an embodiment with a model, under one :class:`RunConfig`.
15
+
16
+ The agent does not own the embodiment's lifetime: open and close it with a
17
+ ``with`` block (or explicitly) around one or more ``run`` calls.
18
+ """
19
+
20
+ def __init__(
21
+ self,
22
+ model: ModelProvider,
23
+ embodiment: Embodiment,
24
+ config: RunConfig | None = None,
25
+ *,
26
+ instructions_template: str = DEFAULT_INSTRUCTIONS,
27
+ ) -> None:
28
+ self.config = config or RunConfig()
29
+ self.runtime = Runtime(model, embodiment, instructions_template=instructions_template)
30
+
31
+ def run(self, goal: str, *, cancel: threading.Event | None = None) -> RunResult:
32
+ return self.runtime.run(goal, self.config, cancel=cancel)
nomad/budget.py ADDED
@@ -0,0 +1,135 @@
1
+ """Run budgets and the counters behind ``metrics.json``.
2
+
3
+ :class:`BudgetTracker` is the single place where loop events are counted. The
4
+ runtime records what happened; the tracker decides whether any budget is
5
+ exhausted and produces the final :class:`~nomad.core.run.RunMetrics`.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import time
11
+ from collections.abc import Callable
12
+
13
+ from nomad.core.codes import ActionStatus, ModelResponseStatus, StopReason
14
+ from nomad.core.model_io import ModelResponse
15
+ from nomad.core.results import ActionResult
16
+ from nomad.core.run import RunConfig, RunMetrics
17
+
18
+
19
+ class BudgetTracker:
20
+ """Counts decisions, provider calls, tokens, failures, and elapsed time.
21
+
22
+ Consecutive failures are any decisions that did not end in a completed
23
+ action (provider errors, rejections, execution failures); a completed
24
+ action resets the streak. Token budgets apply only while every response has
25
+ reported usage: once any call omits it, totals become unknown and the
26
+ token budget can no longer be enforced.
27
+ """
28
+
29
+ def __init__(self, config: RunConfig, *, clock: Callable[[], float] = time.monotonic) -> None:
30
+ self._config = config
31
+ self._clock = clock
32
+ self._started = clock()
33
+
34
+ self.decisions = 0
35
+ self.provider_calls = 0
36
+ self.provider_errors = 0
37
+ self.rejections = 0
38
+ self.actions_completed = 0
39
+ self.execution_failures = 0
40
+ self.consecutive_failures = 0
41
+ self.model_latency_s = 0.0
42
+ self.sim_time_s = 0.0
43
+ self._input_tokens: int | None = 0
44
+ self._output_tokens: int | None = 0
45
+
46
+ # --- recording ----------------------------------------------------------------
47
+
48
+ def record_decision(self) -> None:
49
+ self.decisions += 1
50
+
51
+ def record_model_response(self, response: ModelResponse) -> None:
52
+ self.provider_calls += 1
53
+ self.model_latency_s += response.latency_s
54
+ usage = response.usage
55
+ self._input_tokens = _add(self._input_tokens, usage.input_tokens if usage else None)
56
+ self._output_tokens = _add(self._output_tokens, usage.output_tokens if usage else None)
57
+ if response.status is not ModelResponseStatus.OK:
58
+ self.provider_errors += 1
59
+ self.consecutive_failures += 1
60
+
61
+ def record_rejection(self) -> None:
62
+ """A proposal that never became a dispatched action (unparseable or invalid)."""
63
+ self.rejections += 1
64
+ self.consecutive_failures += 1
65
+
66
+ def record_no_progress(self) -> None:
67
+ """A decision that made no progress without failing, e.g. an unconfirmed finish request."""
68
+ self.consecutive_failures += 1
69
+
70
+ def record_result(self, result: ActionResult) -> None:
71
+ if result.status is ActionStatus.COMPLETED:
72
+ self.actions_completed += 1
73
+ self.consecutive_failures = 0
74
+ return
75
+ if result.status is ActionStatus.REJECTED:
76
+ self.rejections += 1
77
+ else:
78
+ self.execution_failures += 1
79
+ self.consecutive_failures += 1
80
+
81
+ def record_sim_time(self, sim_time_s: float) -> None:
82
+ self.sim_time_s = sim_time_s
83
+
84
+ # --- queries ------------------------------------------------------------------
85
+
86
+ @property
87
+ def wall_time_s(self) -> float:
88
+ return self._clock() - self._started
89
+
90
+ @property
91
+ def total_tokens(self) -> int | None:
92
+ if self._input_tokens is None or self._output_tokens is None:
93
+ return None
94
+ return self._input_tokens + self._output_tokens
95
+
96
+ def exhausted(self) -> StopReason | None:
97
+ """The first exhausted budget, or ``None`` if the loop may continue."""
98
+ c = self._config
99
+ if self.decisions >= c.max_decisions:
100
+ return StopReason.MAX_DECISIONS
101
+ if self.wall_time_s >= c.max_wall_time_s:
102
+ return StopReason.MAX_WALL_TIME
103
+ if c.max_sim_time_s is not None and self.sim_time_s >= c.max_sim_time_s:
104
+ return StopReason.MAX_SIM_TIME
105
+ if c.max_provider_calls is not None and self.provider_calls >= c.max_provider_calls:
106
+ return StopReason.MAX_PROVIDER_CALLS
107
+ tokens = self.total_tokens
108
+ if c.max_total_tokens is not None and tokens is not None and tokens >= c.max_total_tokens:
109
+ return StopReason.MAX_TOKENS
110
+ if self.consecutive_failures >= c.max_consecutive_failures:
111
+ return StopReason.MAX_CONSECUTIVE_FAILURES
112
+ return None
113
+
114
+ def metrics(self) -> RunMetrics:
115
+ known = self.provider_calls > 0
116
+ return RunMetrics(
117
+ decisions=self.decisions,
118
+ provider_calls=self.provider_calls,
119
+ provider_errors=self.provider_errors,
120
+ rejections=self.rejections,
121
+ actions_completed=self.actions_completed,
122
+ execution_failures=self.execution_failures,
123
+ input_tokens=self._input_tokens if known else None,
124
+ output_tokens=self._output_tokens if known else None,
125
+ model_latency_s=self.model_latency_s,
126
+ wall_time_s=max(0.0, self.wall_time_s),
127
+ sim_time_s=self.sim_time_s,
128
+ )
129
+
130
+
131
+ def _add(total: int | None, value: int | None) -> int | None:
132
+ # Unknown is sticky: a sum that includes an unreported value is unknown.
133
+ if total is None or value is None:
134
+ return None
135
+ return total + value
nomad/context.py ADDED
@@ -0,0 +1,89 @@
1
+ """Builds the canonical, provider-agnostic context for each model decision."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import deque
6
+ from typing import Any
7
+
8
+ from nomad.core.codes import FeedbackCode, ObservationMode
9
+ from nomad.core.manifest import EmbodimentManifest, proposal_json_schema
10
+ from nomad.core.model_io import AgentContext, HistoryEntry
11
+ from nomad.core.observations import Observation
12
+
13
+ DEFAULT_INSTRUCTIONS = """\
14
+ You control a robot embodiment ({embodiment_id}: {robot}, task {task}) one action at a time.
15
+ After each action you receive a new observation and feedback on what actually executed.
16
+
17
+ Conventions:
18
+ - Poses are in the right-handed '{pose_frame}' frame: meters, radians, quaternions as [x, y, z, w].
19
+ - move_ee: translation_m is the TOTAL end-effector displacement for this action, not a velocity.
20
+ rotation_vector_rad is an axis-angle vector (its length is the angle), applied in the frame.
21
+ It is not Euler roll/pitch/yaw.
22
+ - set_gripper: closure 0 is fully open, 1 is fully closed. It is a setting, not a force.
23
+ - duration_s is the maximum execution time; motion that does not converge reports a timeout.
24
+
25
+ Rules:
26
+ - Respond with exactly one action per decision, matching the provided schema.
27
+ - Requests outside the limits are rejected, not clipped; read the feedback and adjust.
28
+ - Set request_finish to true when you believe the task is complete. An independent check
29
+ decides success; requesting finish does not make the task succeed.
30
+ """
31
+
32
+
33
+ class ContextBuilder:
34
+ """Turns observations and feedback into :class:`AgentContext` records.
35
+
36
+ The builder is the only place where model-visible state is decided:
37
+
38
+ - ``privileged`` simulator state is removed unless ``mode`` is
39
+ ``privileged_state``;
40
+ - history keeps the last ``history_window`` decisions; a window of 0
41
+ disables both history and last-error feedback (the feedback-off ablation);
42
+ - ``last_error`` is the message of the most recent unsuccessful decision and
43
+ clears after a successful action.
44
+ """
45
+
46
+ def __init__(
47
+ self,
48
+ manifest: EmbodimentManifest,
49
+ *,
50
+ mode: ObservationMode,
51
+ history_window: int,
52
+ instructions_template: str = DEFAULT_INSTRUCTIONS,
53
+ ) -> None:
54
+ self._mode = mode
55
+ self._feedback_enabled = history_window > 0
56
+ self._history: deque[HistoryEntry] = deque(maxlen=history_window)
57
+ self._last_error: str | None = None
58
+ self.schema: dict[str, Any] = proposal_json_schema(manifest)
59
+ self.instructions = instructions_template.format(
60
+ embodiment_id=manifest.embodiment_id,
61
+ robot=manifest.robot,
62
+ task=manifest.task,
63
+ pose_frame=manifest.pose_frame,
64
+ )
65
+
66
+ def record(self, entry: HistoryEntry) -> None:
67
+ self._history.append(entry)
68
+ if entry.code is FeedbackCode.OK:
69
+ self._last_error = None
70
+ else:
71
+ detail = f": {entry.message}" if entry.message else ""
72
+ self._last_error = f"step {entry.step}: {entry.code}{detail}"
73
+
74
+ def build(self, goal: str, observation: Observation) -> AgentContext:
75
+ visible = observation
76
+ if (
77
+ self._mode is not ObservationMode.PRIVILEGED_STATE
78
+ and observation.privileged is not None
79
+ ):
80
+ visible = observation.model_copy(update={"privileged": None})
81
+ return AgentContext(
82
+ goal=goal,
83
+ instructions=self.instructions,
84
+ mode=self._mode,
85
+ observation=visible,
86
+ action_schema=self.schema,
87
+ history=tuple(self._history),
88
+ last_error=self._last_error if self._feedback_enabled else None,
89
+ )
nomad/core/__init__.py ADDED
@@ -0,0 +1,78 @@
1
+ """Typed contracts and protocols shared by the runtime and every adapter."""
2
+
3
+ from nomad.core.actions import (
4
+ Action,
5
+ ActionCall,
6
+ MoveEEArgs,
7
+ MoveEECall,
8
+ Proposal,
9
+ SetGripperArgs,
10
+ SetGripperCall,
11
+ )
12
+ from nomad.core.codes import (
13
+ ActionStatus,
14
+ FeedbackCode,
15
+ ModelResponseStatus,
16
+ ObservationMode,
17
+ OutcomeStatus,
18
+ RunStatus,
19
+ StopReason,
20
+ )
21
+ from nomad.core.errors import ManifestError, MissingDependencyError, ModelError, NomadError
22
+ from nomad.core.manifest import (
23
+ EmbodimentManifest,
24
+ load_manifest,
25
+ parse_manifest,
26
+ proposal_json_schema,
27
+ )
28
+ from nomad.core.model_io import AgentContext, HistoryEntry, ModelResponse, Usage
29
+ from nomad.core.observations import CameraFrame, CameraIntrinsics, Observation, RobotState
30
+ from nomad.core.protocols import DescribesRunConfig, Embodiment, ModelProvider, SupportsVideo
31
+ from nomad.core.results import ActionResult, TaskOutcome
32
+ from nomad.core.run import RunConfig, RunMetrics, RunResult
33
+ from nomad.core.types import PROTOCOL_VERSION, Pose, Record
34
+
35
+ __all__ = [
36
+ "PROTOCOL_VERSION",
37
+ "Action",
38
+ "ActionCall",
39
+ "ActionResult",
40
+ "ActionStatus",
41
+ "AgentContext",
42
+ "CameraFrame",
43
+ "CameraIntrinsics",
44
+ "DescribesRunConfig",
45
+ "Embodiment",
46
+ "EmbodimentManifest",
47
+ "FeedbackCode",
48
+ "HistoryEntry",
49
+ "ManifestError",
50
+ "MissingDependencyError",
51
+ "ModelError",
52
+ "ModelProvider",
53
+ "ModelResponse",
54
+ "ModelResponseStatus",
55
+ "MoveEEArgs",
56
+ "MoveEECall",
57
+ "NomadError",
58
+ "Observation",
59
+ "ObservationMode",
60
+ "OutcomeStatus",
61
+ "Pose",
62
+ "Proposal",
63
+ "Record",
64
+ "RobotState",
65
+ "RunConfig",
66
+ "RunMetrics",
67
+ "RunResult",
68
+ "RunStatus",
69
+ "SetGripperArgs",
70
+ "SetGripperCall",
71
+ "StopReason",
72
+ "SupportsVideo",
73
+ "TaskOutcome",
74
+ "Usage",
75
+ "load_manifest",
76
+ "parse_manifest",
77
+ "proposal_json_schema",
78
+ ]
nomad/core/actions.py ADDED
@@ -0,0 +1,88 @@
1
+ """Action vocabulary: what a model may propose and what the runtime dispatches.
2
+
3
+ A model produces a :class:`Proposal` of :data:`ActionCall` items (name plus typed
4
+ arguments). Only the runtime wraps a call into an :class:`Action`, attaching the
5
+ execution identifiers and freshness deadline the model must not control.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import Annotated, Literal, TypeAlias
11
+
12
+ from pydantic import Field, model_validator
13
+
14
+ from nomad.core.types import (
15
+ FiniteFloat,
16
+ PositiveFloat,
17
+ Record,
18
+ UnitInterval,
19
+ Vec3,
20
+ )
21
+
22
+
23
+ class MoveEEArgs(Record):
24
+ """Total end-effector displacement in a declared frame, bounded in duration."""
25
+
26
+ frame: Annotated[str, Field(min_length=1)]
27
+ translation_m: Vec3
28
+ rotation_vector_rad: Vec3
29
+ duration_s: PositiveFloat
30
+
31
+
32
+ class SetGripperArgs(Record):
33
+ """Desired closure: 0 is fully open, 1 is fully closed. Not a force command."""
34
+
35
+ closure: UnitInterval
36
+ duration_s: PositiveFloat
37
+
38
+
39
+ class MoveEECall(Record):
40
+ name: Literal["move_ee"]
41
+ arguments: MoveEEArgs
42
+
43
+
44
+ class SetGripperCall(Record):
45
+ name: Literal["set_gripper"]
46
+ arguments: SetGripperArgs
47
+
48
+
49
+ ActionCall: TypeAlias = Annotated[MoveEECall | SetGripperCall, Field(discriminator="name")]
50
+
51
+
52
+ class Proposal(Record):
53
+ """One model decision.
54
+
55
+ v0.1 rule: exactly one action unless ``request_finish`` is set, in which case
56
+ zero or one action is allowed. A finish request only triggers an independent
57
+ outcome check; it never marks the task successful by itself.
58
+ """
59
+
60
+ schema_version: Literal["0.1"]
61
+ actions: tuple[ActionCall, ...]
62
+ request_finish: bool
63
+ rationale: Annotated[str, Field(max_length=500)] | None = None
64
+
65
+ @model_validator(mode="after")
66
+ def _action_count(self) -> Proposal:
67
+ n = len(self.actions)
68
+ if self.request_finish and n > 1:
69
+ raise ValueError("a finish request may include at most one action")
70
+ if not self.request_finish and n != 1:
71
+ raise ValueError(f"exactly one action is required without request_finish (got {n})")
72
+ return self
73
+
74
+
75
+ class Action(Record):
76
+ """Runtime execution envelope around one validated call."""
77
+
78
+ call: ActionCall
79
+ episode_id: str
80
+ action_id: str
81
+ observation_id: str
82
+ manifest_version: str
83
+ deadline_wall_s: FiniteFloat
84
+ protocol_version: Literal["0.1"] = "0.1"
85
+
86
+ @property
87
+ def name(self) -> str:
88
+ return self.call.name
nomad/core/codes.py ADDED
@@ -0,0 +1,103 @@
1
+ """Status and feedback vocabularies shared by the runtime, adapters, and traces.
2
+
3
+ Every enum is a ``str`` enum so values serialize as plain strings in JSON
4
+ traces and can be compared against literals in tests and analysis scripts.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from enum import Enum
10
+
11
+
12
+ class _StrEnum(str, Enum):
13
+ def __str__(self) -> str:
14
+ return str(self.value)
15
+
16
+
17
+ class ActionStatus(_StrEnum):
18
+ """What happened to one dispatched (or refused) action."""
19
+
20
+ COMPLETED = "completed"
21
+ REJECTED = "rejected"
22
+ FAILED = "failed"
23
+ TIMED_OUT = "timed_out"
24
+ CANCELLED = "cancelled"
25
+
26
+
27
+ class FeedbackCode(_StrEnum):
28
+ """Machine-readable reason attached to every action result and rejection."""
29
+
30
+ OK = "ok"
31
+
32
+ # Structural problems with a model proposal.
33
+ MALFORMED_PROPOSAL = "malformed_proposal"
34
+ UNKNOWN_ACTION = "unknown_action"
35
+ UNKNOWN_FIELD = "unknown_field"
36
+ NONFINITE_VALUE = "nonfinite_value"
37
+ INVALID_DIMENSION = "invalid_dimension"
38
+ OUT_OF_RANGE = "out_of_range"
39
+
40
+ # Semantic problems relative to the active embodiment and observation.
41
+ UNSUPPORTED_ACTION = "unsupported_action"
42
+ UNSUPPORTED_FRAME = "unsupported_frame"
43
+ TRANSLATION_LIMIT = "translation_limit"
44
+ ROTATION_LIMIT = "rotation_limit"
45
+ DURATION_LIMIT = "duration_limit"
46
+ WORKSPACE_VIOLATION = "workspace_violation"
47
+ STALE_OBSERVATION = "stale_observation"
48
+ MANIFEST_MISMATCH = "manifest_mismatch"
49
+
50
+ # Execution outcomes reported by an embodiment.
51
+ TARGET_NOT_REACHED = "target_not_reached"
52
+ EXECUTION_INTERRUPTED = "execution_interrupted"
53
+
54
+ # Model-provider and loop-level feedback.
55
+ PROVIDER_ERROR = "provider_error"
56
+ PROVIDER_REFUSAL = "provider_refusal"
57
+ FINISH_NOT_CONFIRMED = "finish_not_confirmed"
58
+ CANCELLED = "cancelled"
59
+
60
+
61
+ class OutcomeStatus(_StrEnum):
62
+ """Task outcome as judged by the embodiment's evaluator, never by the model."""
63
+
64
+ SUCCESS = "success"
65
+ FAILURE = "failure"
66
+ ONGOING = "ongoing"
67
+ UNKNOWN = "unknown"
68
+
69
+
70
+ class ModelResponseStatus(_StrEnum):
71
+ OK = "ok"
72
+ ERROR = "error"
73
+ REFUSAL = "refusal"
74
+
75
+
76
+ class ObservationMode(_StrEnum):
77
+ """Which state the model may see. Results from the two modes are reported separately."""
78
+
79
+ RGB_PROPRIO = "rgb_proprio"
80
+ PRIVILEGED_STATE = "privileged_state"
81
+
82
+
83
+ class RunStatus(_StrEnum):
84
+ SUCCESS = "success"
85
+ FAILURE = "failure"
86
+ BUDGET_EXHAUSTED = "budget_exhausted"
87
+ CANCELLED = "cancelled"
88
+ ERROR = "error"
89
+
90
+
91
+ class StopReason(_StrEnum):
92
+ """Why the runtime loop ended."""
93
+
94
+ TASK_SUCCESS = "task_success"
95
+ TASK_FAILURE = "task_failure"
96
+ MAX_DECISIONS = "max_decisions"
97
+ MAX_WALL_TIME = "max_wall_time"
98
+ MAX_SIM_TIME = "max_sim_time"
99
+ MAX_PROVIDER_CALLS = "max_provider_calls"
100
+ MAX_TOKENS = "max_tokens"
101
+ MAX_CONSECUTIVE_FAILURES = "max_consecutive_failures"
102
+ CANCELLED = "cancelled"
103
+ ERROR = "error"
nomad/core/errors.py ADDED
@@ -0,0 +1,19 @@
1
+ """Exception types. Bad model output is reported as feedback, not raised."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class NomadError(Exception):
7
+ """Base class for errors raised by Nomad itself."""
8
+
9
+
10
+ class ManifestError(NomadError):
11
+ """An embodiment manifest could not be loaded or is inconsistent."""
12
+
13
+
14
+ class ModelError(NomadError):
15
+ """A model provider failed in a way its adapter could not normalize."""
16
+
17
+
18
+ class MissingDependencyError(NomadError, ImportError):
19
+ """An optional integration was requested without its dependencies installed."""