nomad-harness 0.1.0.dev0__tar.gz → 0.1.0.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/PKG-INFO +38 -13
  2. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/README.md +35 -11
  3. nomad_harness-0.1.0.dev1/examples/quickstart.py +24 -0
  4. nomad_harness-0.1.0.dev1/examples/robosuite_openai_lift.py +104 -0
  5. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/pyproject.toml +6 -2
  6. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/__init__.py +1 -1
  7. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/context.py +44 -3
  8. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/model_io.py +6 -0
  9. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/observations.py +5 -0
  10. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/_imports.py +2 -1
  11. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/adapter.py +1 -0
  12. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/config.py +2 -1
  13. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/models/__init__.py +2 -1
  14. nomad_harness-0.1.0.dev1/src/nomad/models/_encoding.py +205 -0
  15. nomad_harness-0.1.0.dev1/src/nomad/models/_tools.py +113 -0
  16. nomad_harness-0.1.0.dev1/src/nomad/models/openai/__init__.py +10 -0
  17. nomad_harness-0.1.0.dev1/src/nomad/models/openai/_imports.py +20 -0
  18. nomad_harness-0.1.0.dev1/src/nomad/models/openai/adapter.py +367 -0
  19. nomad_harness-0.1.0.dev1/src/nomad/models/openai/config.py +48 -0
  20. nomad_harness-0.1.0.dev1/src/nomad/models/openai/schema.py +111 -0
  21. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/validation.py +10 -2
  22. nomad_harness-0.1.0.dev1/tests/fixtures/openai/bad_request_error.json +10 -0
  23. nomad_harness-0.1.0.dev1/tests/fixtures/openai/function_call.json +197 -0
  24. nomad_harness-0.1.0.dev1/tests/fixtures/openai/incomplete.json +199 -0
  25. nomad_harness-0.1.0.dev1/tests/fixtures/openai/robosuite_lift_seed0_front_rgb.png +0 -0
  26. nomad_harness-0.1.0.dev1/tests/fixtures/openai/robosuite_lift_seed0_observation.json +89 -0
  27. nomad_harness-0.1.0.dev1/tests/fixtures/openai/structured_output.json +240 -0
  28. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_context.py +26 -1
  29. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_import_hygiene.py +15 -0
  30. nomad_harness-0.1.0.dev1/tests/test_model_encoding.py +175 -0
  31. nomad_harness-0.1.0.dev1/tests/test_model_tools.py +140 -0
  32. nomad_harness-0.1.0.dev1/tests/test_openai_adapter.py +390 -0
  33. nomad_harness-0.1.0.dev1/tests/test_openai_live.py +87 -0
  34. nomad_harness-0.1.0.dev1/tests/test_openai_schema.py +168 -0
  35. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_records.py +10 -0
  36. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_robosuite_config.py +4 -2
  37. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_robosuite_embodiment.py +11 -0
  38. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_runtime.py +13 -0
  39. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_validation.py +9 -0
  40. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/.github/workflows/ci.yml +0 -0
  41. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/.github/workflows/release.yml +0 -0
  42. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/.gitignore +0 -0
  43. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/LICENSE +0 -0
  44. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/examples/offline_demo.py +0 -0
  45. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/examples/robosuite_scripted_lift.py +0 -0
  46. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/agent.py +0 -0
  47. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/budget.py +0 -0
  48. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/__init__.py +0 -0
  49. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/actions.py +0 -0
  50. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/codes.py +0 -0
  51. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/errors.py +0 -0
  52. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/geometry.py +0 -0
  53. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/manifest.py +0 -0
  54. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/protocols.py +0 -0
  55. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/results.py +0 -0
  56. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/run.py +0 -0
  57. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/core/types.py +0 -0
  58. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/__init__.py +0 -0
  59. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/_execution.py +0 -0
  60. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/_manifests.py +0 -0
  61. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/_png.py +0 -0
  62. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/fake.py +0 -0
  63. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/manifests/fake.yaml +0 -0
  64. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/manifests/robosuite_lift.yaml +0 -0
  65. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/__init__.py +0 -0
  66. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/conversions.py +0 -0
  67. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/embodiments/robosuite/video.py +0 -0
  68. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/models/fake.py +0 -0
  69. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/models/scripted.py +0 -0
  70. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/py.typed +0 -0
  71. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/runtime.py +0 -0
  72. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/src/nomad/tracing.py +0 -0
  73. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/support.py +0 -0
  74. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_actions.py +0 -0
  75. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_budget.py +0 -0
  76. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_execution_guard.py +0 -0
  77. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_fake_embodiment.py +0 -0
  78. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_fake_model.py +0 -0
  79. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_geometry.py +0 -0
  80. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_manifest.py +0 -0
  81. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_offline_demo.py +0 -0
  82. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_protocols.py +0 -0
  83. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_robosuite_conversions.py +0 -0
  84. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_robosuite_execution.py +0 -0
  85. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_robosuite_scripted_lift.py +0 -0
  86. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_scripted_policy.py +0 -0
  87. {nomad_harness-0.1.0.dev0 → nomad_harness-0.1.0.dev1}/tests/test_tracing.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: nomad-harness
3
- Version: 0.1.0.dev0
3
+ Version: 0.1.0.dev1
4
4
  Summary: Action runtime connecting frontier models to simulated and physical embodiments.
5
5
  Project-URL: Repository, https://github.com/Lambda-Robotics-Inc/Nomad_Harness_Interface
6
6
  Project-URL: Issues, https://github.com/Lambda-Robotics-Inc/Nomad_Harness_Interface/issues
@@ -24,13 +24,14 @@ Requires-Dist: anthropic; extra == 'anthropic'
24
24
  Provides-Extra: dev
25
25
  Requires-Dist: jsonschema>=4.18; extra == 'dev'
26
26
  Requires-Dist: mypy>=1.10; extra == 'dev'
27
+ Requires-Dist: openai<4,>=3.24; extra == 'dev'
27
28
  Requires-Dist: pytest>=8.0; extra == 'dev'
28
29
  Requires-Dist: ruff>=0.5; extra == 'dev'
29
30
  Requires-Dist: types-pyyaml; extra == 'dev'
30
31
  Provides-Extra: gemini
31
32
  Requires-Dist: google-genai; extra == 'gemini'
32
33
  Provides-Extra: openai
33
- Requires-Dist: openai; extra == 'openai'
34
+ Requires-Dist: openai<4,>=3.24; extra == 'openai'
34
35
  Provides-Extra: robosuite
35
36
  Requires-Dist: imageio[ffmpeg]; extra == 'robosuite'
36
37
  Requires-Dist: mujoco==3.3.7; extra == 'robosuite'
@@ -44,7 +45,7 @@ Description-Content-Type: text/markdown
44
45
 
45
46
  Nomad connects a model provider to a simulator or robot adapter through explicit observation, action, and execution-feedback contracts. The first milestone is a working simulation package: change the model or simulation backend without rewriting the agent loop.
46
47
 
47
- **Status: implementation brief, Day 2 complete.** This README defines the work to build v0.1. The core contracts, offline runtime, and fake backend/model from [the first engineering task](#the-first-engineering-task) are implemented and tested, as is the robosuite `Lift` adapter with its privileged-state scripted baseline (Day 2 of [the delivery plan](#8-seven-day-delivery-plan)). Everything else below (provider adapters, Isaac Lab, CLI, evaluation runner) is still a proposed contract, not implemented functionality. No package release, benchmark result, or hardware support is claimed here.
48
+ **Status: implementation brief, Day 3 complete.** This README defines the work to build v0.1. The core contracts, offline runtime, and fake backend/model from [the first engineering task](#the-first-engineering-task) are implemented and tested, as are the robosuite `Lift` adapter with its privileged-state scripted baseline (Day 2 of [the delivery plan](#8-seven-day-delivery-plan)) and the OpenAI provider adapter with a model-driven robosuite rollout (Day 3). Everything else below (Anthropic and Gemini adapters, Isaac Lab, CLI, evaluation runner) is still a proposed contract, not implemented functionality. Development pre-releases are on PyPI as [`nomad-harness`](https://pypi.org/project/nomad-harness/) (install with `pip install --pre nomad-harness`); no benchmark result or hardware support is claimed here.
48
49
 
49
50
  **For the implementation partner:** start with [the first engineering task](#the-first-engineering-task), follow [the seven-day delivery plan](#8-seven-day-delivery-plan), and use [the release acceptance checklist](#10-release-acceptance-checklist) as the definition of done.
50
51
 
@@ -56,7 +57,8 @@ Nomad connects a model provider to a simulator or robot adapter through explicit
56
57
  | `FakeEmbodiment` (deterministic cube lift) and `FakeModel` (scripted) | Implemented, offline tests |
57
58
  | `Robosuite` adapter: Panda `Lift`, absolute base-frame OSC, RGB/proprioception, video | Implemented, simulator tests (robosuite 1.5.2, mujoco 3.3.7) |
58
59
  | `ScriptedLiftPolicy` baseline (privileged object state) | Implemented; succeeds on robosuite `Lift` and the fake world |
59
- | Isaac Lab, OpenAI, Anthropic, Gemini adapters; `nomad` CLI; evaluation runner | Not started |
60
+ | `OpenAI` adapter: Responses API, strict tool calling, RGB + proprioception context | Implemented; offline tests on recorded responses, opt-in live test (openai 3.24) |
61
+ | Isaac Lab, Anthropic, Gemini adapters; `nomad` CLI; evaluation runner | Not started |
60
62
 
61
63
  Run the offline demo (Python 3.10+; no API keys, simulator, or GPU):
62
64
 
@@ -76,26 +78,42 @@ pytest -m robosuite # adapter acceptance test
76
78
 
77
79
  The baseline reads the simulator's cube position, so its runs are labeled `privileged_state` and are not comparable with model runs that see only images and proprioception. On headless Linux, set `MUJOCO_GL=egl` or `MUJOCO_GL=osmesa` for offscreen rendering.
78
80
 
81
+ Run an OpenAI model on robosuite `Lift` (spends API credits). The model sees the front camera image and robot state only (`rgb_proprio`); success comes from robosuite's own predicate. The key is read from `OPENAI_API_KEY` and never written to traces; the model must be one your account can access.
82
+
83
+ ```bash
84
+ python -m pip install -e ".[dev,robosuite,openai]"
85
+ export OPENAI_API_KEY=... # better: source it from a private file outside the repo
86
+ export NOMAD_OPENAI_MODEL=your-model-id
87
+ python examples/quickstart.py # the minimal script shown in section 1
88
+ python examples/robosuite_openai_lift.py --seed 0 # with options; add --reasoning-effort high
89
+ NOMAD_OPENAI_LIVE=1 pytest -m openai_live # live multimodal proposal test (two calls)
90
+ ```
91
+
92
+ Plain `pytest` and CI never call the API: the adapter is tested offline against responses recorded from the live API, and the live test runs only with `NOMAD_OPENAI_LIVE=1`.
93
+
94
+ **Pilot, not a benchmark.** On 2026-10-06, `gpt-6.1-sol` (default reasoning effort, history window 8, 30-decision budget) lifted the cube on seeds 0–3 in 18, 9, 8, and 14 decisions; seed 4 exhausted the budget after re-grasping six times at the same position. Rerun with `--reasoning-effort high`, seed 4 succeeded in 8 decisions. That is five seeds plus one rerun: too few for a success rate or for any claim about reasoning effort. The [experiment matrix](#initial-experiment-matrix) calls for at least 10 seeds per configuration.
95
+
79
96
  ## 1. Product goal and first release
80
97
 
81
- The developer experience should eventually be:
98
+ The developer experience is this complete script ([`examples/quickstart.py`](examples/quickstart.py)), implemented for OpenAI and robosuite:
82
99
 
83
100
  ```python
84
- # Target API: implement this behavior before advertising it as a quickstart.
85
101
  import os
86
102
 
87
103
  from nomad import Agent, RunConfig
88
- from nomad.models import OpenAI
89
104
  from nomad.embodiments import Robosuite
105
+ from nomad.models import OpenAI
90
106
 
91
107
  with Robosuite(task="Lift", robot="Panda") as env:
92
108
  agent = Agent(
93
- model=OpenAI(model=os.environ["NOMAD_MODEL"]),
109
+ model=OpenAI(model=os.environ["NOMAD_OPENAI_MODEL"]),
94
110
  embodiment=env,
95
- config=RunConfig(max_decisions=30, max_wall_time_s=180),
111
+ config=RunConfig(max_decisions=30, max_wall_time_s=600),
96
112
  )
97
113
  result = agent.run("Lift the cube off the table.")
98
- print(result.status, result.trace_dir)
114
+
115
+ print(result.status, result.outcome.reason)
116
+ print("trace:", result.trace_dir)
99
117
  ```
100
118
 
101
119
  The provider reads its API key from the environment. Model identifiers are supplied by the user and checked against actual account access; do not hard-code an assumed future model name.
@@ -339,7 +357,14 @@ See the official [environment registry](https://isaac-sim.github.io/IsaacLab/mai
339
357
 
340
358
  ### Model providers
341
359
 
342
- Implement OpenAI first, then Anthropic and Gemini. Each adapter must:
360
+ Implement OpenAI first, then Anthropic and Gemini. OpenAI is implemented (`nomad.models.OpenAI`); its shared pieces are meant to be reused by the others:
361
+
362
+ - `nomad.models._tools` turns the proposal schema into one strict tool per action plus `finish`, and turns tool calls back into an unvalidated raw proposal. Flat per-action schemas fit every provider's strict-schema subset better than one schema with actions in an `anyOf`.
363
+ - `nomad.models._encoding` renders the context as system text, user text, and images, with camera poses in the robot-state frame and an image-free summary for traces.
364
+ - Each provider adds only a schema converter for its JSON Schema dialect (`nomad.models.openai.schema`) and a thin adapter. The runtime keeps the loop: provider agent loops and automatic tool runners are not used.
365
+ - An `OpenAICompatible` adapter over Chat Completions (vLLM, Ollama, OpenRouter) can reuse the same pieces later.
366
+
367
+ Each adapter must:
343
368
 
344
369
  - Accept a user-provided model identifier, key through environment configuration, a timeout, and an explicit call budget.
345
370
  - Encode the same canonical context into the provider's image/text request format.
@@ -509,9 +534,9 @@ A second simulator alone is not evidence of universal embodiment transfer. Add a
509
534
 
510
535
  - [ ] A fresh checkout can run the offline example using documented commands.
511
536
  - [ ] Importing `nomad` requires no simulator installation and starts no application.
512
- - [ ] robosuite scripted and model-driven episodes save valid traces and outcomes.
537
+ - [x] robosuite scripted and model-driven episodes save valid traces and outcomes.
513
538
  - [ ] Isaac Lab scripted and model-driven episodes use the same runtime contract, if listed as supported.
514
- - [ ] Each advertised provider passes a live multimodal structured-proposal test using an accessible model.
539
+ - [ ] Each advertised provider passes a live multimodal structured-proposal test using an accessible model. (OpenAI passes; Anthropic and Gemini pending.)
515
540
  - [ ] Adapter tests confirm units, frames, rotation composition, gripper mapping, and total-displacement semantics.
516
541
  - [ ] Invalid and stale commands cannot reach execution; timeouts and partial failures have explicit outcomes.
517
542
  - [ ] Task success comes from a defined evaluator, and model-visible privileged state is labeled.
@@ -4,7 +4,7 @@
4
4
 
5
5
  Nomad connects a model provider to a simulator or robot adapter through explicit observation, action, and execution-feedback contracts. The first milestone is a working simulation package: change the model or simulation backend without rewriting the agent loop.
6
6
 
7
- **Status: implementation brief, Day 2 complete.** This README defines the work to build v0.1. The core contracts, offline runtime, and fake backend/model from [the first engineering task](#the-first-engineering-task) are implemented and tested, as is the robosuite `Lift` adapter with its privileged-state scripted baseline (Day 2 of [the delivery plan](#8-seven-day-delivery-plan)). Everything else below (provider adapters, Isaac Lab, CLI, evaluation runner) is still a proposed contract, not implemented functionality. No package release, benchmark result, or hardware support is claimed here.
7
+ **Status: implementation brief, Day 3 complete.** This README defines the work to build v0.1. The core contracts, offline runtime, and fake backend/model from [the first engineering task](#the-first-engineering-task) are implemented and tested, as are the robosuite `Lift` adapter with its privileged-state scripted baseline (Day 2 of [the delivery plan](#8-seven-day-delivery-plan)) and the OpenAI provider adapter with a model-driven robosuite rollout (Day 3). Everything else below (Anthropic and Gemini adapters, Isaac Lab, CLI, evaluation runner) is still a proposed contract, not implemented functionality. Development pre-releases are on PyPI as [`nomad-harness`](https://pypi.org/project/nomad-harness/) (install with `pip install --pre nomad-harness`); no benchmark result or hardware support is claimed here.
8
8
 
9
9
  **For the implementation partner:** start with [the first engineering task](#the-first-engineering-task), follow [the seven-day delivery plan](#8-seven-day-delivery-plan), and use [the release acceptance checklist](#10-release-acceptance-checklist) as the definition of done.
10
10
 
@@ -16,7 +16,8 @@ Nomad connects a model provider to a simulator or robot adapter through explicit
16
16
  | `FakeEmbodiment` (deterministic cube lift) and `FakeModel` (scripted) | Implemented, offline tests |
17
17
  | `Robosuite` adapter: Panda `Lift`, absolute base-frame OSC, RGB/proprioception, video | Implemented, simulator tests (robosuite 1.5.2, mujoco 3.3.7) |
18
18
  | `ScriptedLiftPolicy` baseline (privileged object state) | Implemented; succeeds on robosuite `Lift` and the fake world |
19
- | Isaac Lab, OpenAI, Anthropic, Gemini adapters; `nomad` CLI; evaluation runner | Not started |
19
+ | `OpenAI` adapter: Responses API, strict tool calling, RGB + proprioception context | Implemented; offline tests on recorded responses, opt-in live test (openai 3.24) |
20
+ | Isaac Lab, Anthropic, Gemini adapters; `nomad` CLI; evaluation runner | Not started |
20
21
 
21
22
  Run the offline demo (Python 3.10+; no API keys, simulator, or GPU):
22
23
 
@@ -36,26 +37,42 @@ pytest -m robosuite # adapter acceptance test
36
37
 
37
38
  The baseline reads the simulator's cube position, so its runs are labeled `privileged_state` and are not comparable with model runs that see only images and proprioception. On headless Linux, set `MUJOCO_GL=egl` or `MUJOCO_GL=osmesa` for offscreen rendering.
38
39
 
40
+ Run an OpenAI model on robosuite `Lift` (spends API credits). The model sees the front camera image and robot state only (`rgb_proprio`); success comes from robosuite's own predicate. The key is read from `OPENAI_API_KEY` and never written to traces; the model must be one your account can access.
41
+
42
+ ```bash
43
+ python -m pip install -e ".[dev,robosuite,openai]"
44
+ export OPENAI_API_KEY=... # better: source it from a private file outside the repo
45
+ export NOMAD_OPENAI_MODEL=your-model-id
46
+ python examples/quickstart.py # the minimal script shown in section 1
47
+ python examples/robosuite_openai_lift.py --seed 0 # with options; add --reasoning-effort high
48
+ NOMAD_OPENAI_LIVE=1 pytest -m openai_live # live multimodal proposal test (two calls)
49
+ ```
50
+
51
+ Plain `pytest` and CI never call the API: the adapter is tested offline against responses recorded from the live API, and the live test runs only with `NOMAD_OPENAI_LIVE=1`.
52
+
53
+ **Pilot, not a benchmark.** On 2026-10-06, `gpt-6.1-sol` (default reasoning effort, history window 8, 30-decision budget) lifted the cube on seeds 0–3 in 18, 9, 8, and 14 decisions; seed 4 exhausted the budget after re-grasping six times at the same position. Rerun with `--reasoning-effort high`, seed 4 succeeded in 8 decisions. That is five seeds plus one rerun: too few for a success rate or for any claim about reasoning effort. The [experiment matrix](#initial-experiment-matrix) calls for at least 10 seeds per configuration.
54
+
39
55
  ## 1. Product goal and first release
40
56
 
41
- The developer experience should eventually be:
57
+ The developer experience is this complete script ([`examples/quickstart.py`](examples/quickstart.py)), implemented for OpenAI and robosuite:
42
58
 
43
59
  ```python
44
- # Target API: implement this behavior before advertising it as a quickstart.
45
60
  import os
46
61
 
47
62
  from nomad import Agent, RunConfig
48
- from nomad.models import OpenAI
49
63
  from nomad.embodiments import Robosuite
64
+ from nomad.models import OpenAI
50
65
 
51
66
  with Robosuite(task="Lift", robot="Panda") as env:
52
67
  agent = Agent(
53
- model=OpenAI(model=os.environ["NOMAD_MODEL"]),
68
+ model=OpenAI(model=os.environ["NOMAD_OPENAI_MODEL"]),
54
69
  embodiment=env,
55
- config=RunConfig(max_decisions=30, max_wall_time_s=180),
70
+ config=RunConfig(max_decisions=30, max_wall_time_s=600),
56
71
  )
57
72
  result = agent.run("Lift the cube off the table.")
58
- print(result.status, result.trace_dir)
73
+
74
+ print(result.status, result.outcome.reason)
75
+ print("trace:", result.trace_dir)
59
76
  ```
60
77
 
61
78
  The provider reads its API key from the environment. Model identifiers are supplied by the user and checked against actual account access; do not hard-code an assumed future model name.
@@ -299,7 +316,14 @@ See the official [environment registry](https://isaac-sim.github.io/IsaacLab/mai
299
316
 
300
317
  ### Model providers
301
318
 
302
- Implement OpenAI first, then Anthropic and Gemini. Each adapter must:
319
+ Implement OpenAI first, then Anthropic and Gemini. OpenAI is implemented (`nomad.models.OpenAI`); its shared pieces are meant to be reused by the others:
320
+
321
+ - `nomad.models._tools` turns the proposal schema into one strict tool per action plus `finish`, and turns tool calls back into an unvalidated raw proposal. Flat per-action schemas fit every provider's strict-schema subset better than one schema with actions in an `anyOf`.
322
+ - `nomad.models._encoding` renders the context as system text, user text, and images, with camera poses in the robot-state frame and an image-free summary for traces.
323
+ - Each provider adds only a schema converter for its JSON Schema dialect (`nomad.models.openai.schema`) and a thin adapter. The runtime keeps the loop: provider agent loops and automatic tool runners are not used.
324
+ - An `OpenAICompatible` adapter over Chat Completions (vLLM, Ollama, OpenRouter) can reuse the same pieces later.
325
+
326
+ Each adapter must:
303
327
 
304
328
  - Accept a user-provided model identifier, key through environment configuration, a timeout, and an explicit call budget.
305
329
  - Encode the same canonical context into the provider's image/text request format.
@@ -469,9 +493,9 @@ A second simulator alone is not evidence of universal embodiment transfer. Add a
469
493
 
470
494
  - [ ] A fresh checkout can run the offline example using documented commands.
471
495
  - [ ] Importing `nomad` requires no simulator installation and starts no application.
472
- - [ ] robosuite scripted and model-driven episodes save valid traces and outcomes.
496
+ - [x] robosuite scripted and model-driven episodes save valid traces and outcomes.
473
497
  - [ ] Isaac Lab scripted and model-driven episodes use the same runtime contract, if listed as supported.
474
- - [ ] Each advertised provider passes a live multimodal structured-proposal test using an accessible model.
498
+ - [ ] Each advertised provider passes a live multimodal structured-proposal test using an accessible model. (OpenAI passes; Anthropic and Gemini pending.)
475
499
  - [ ] Adapter tests confirm units, frames, rotation composition, gripper mapping, and total-displacement semantics.
476
500
  - [ ] Invalid and stale commands cannot reach execution; timeouts and partial failures have explicit outcomes.
477
501
  - [ ] Task success comes from a defined evaluator, and model-visible privileged state is labeled.
@@ -0,0 +1,24 @@
1
+ """Minimal example: an OpenAI model lifts a cube in robosuite.
2
+
3
+ pip install -e '.[robosuite,openai]'
4
+ export OPENAI_API_KEY=...
5
+ export NOMAD_OPENAI_MODEL=... # a model your account can access
6
+ python examples/quickstart.py
7
+ """
8
+
9
+ import os
10
+
11
+ from nomad import Agent, RunConfig
12
+ from nomad.embodiments import Robosuite
13
+ from nomad.models import OpenAI
14
+
15
+ with Robosuite(task="Lift", robot="Panda") as env:
16
+ agent = Agent(
17
+ model=OpenAI(model=os.environ["NOMAD_OPENAI_MODEL"]),
18
+ embodiment=env,
19
+ config=RunConfig(max_decisions=30, max_wall_time_s=600),
20
+ )
21
+ result = agent.run("Lift the cube off the table.")
22
+
23
+ print(result.status, result.outcome.reason)
24
+ print("trace:", result.trace_dir)
@@ -0,0 +1,104 @@
1
+ """OpenAI model rollout on robosuite Lift: the model sees only images and proprioception.
2
+
3
+ An OpenAI model drives the Panda through the Nomad runtime with canonical
4
+ ``move_ee``/``set_gripper`` actions, one decision at a time, from the front
5
+ camera image and robot state (``rgb_proprio``: no simulator object state).
6
+ Success is decided by robosuite's own Lift predicate, never by the model.
7
+ Every run writes config.json, trace.jsonl, frames/, metrics.json, and video.mp4,
8
+ and its outcome is reported whether or not the cube was lifted.
9
+
10
+ Spends API credits. The model must be one your account can access:
11
+
12
+ pip install -e '.[robosuite,openai]'
13
+ export OPENAI_API_KEY=... # never written to traces
14
+ python examples/robosuite_openai_lift.py --model "$NOMAD_OPENAI_MODEL" [--seed 0]
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import argparse
20
+ import os
21
+ import sys
22
+ from pathlib import Path
23
+
24
+ from nomad import Agent, ObservationMode, RunConfig, RunResult, RunStatus
25
+ from nomad.core import MissingDependencyError, ModelError
26
+ from nomad.embodiments import Robosuite
27
+ from nomad.models import OpenAI
28
+
29
+ GOAL = "Lift the cube off the table."
30
+
31
+
32
+ def print_summary(result: RunResult) -> None:
33
+ m = result.metrics
34
+ tokens = (
35
+ f"{m.input_tokens} in / {m.output_tokens} out"
36
+ if m.input_tokens is not None and m.output_tokens is not None
37
+ else "unknown"
38
+ )
39
+ print(f"status: {result.status} ({result.stop_reason})")
40
+ print(f"outcome: {result.outcome.status}: {result.outcome.reason}")
41
+ print(f"evidence: {result.outcome.evidence}")
42
+ print(
43
+ f"decisions: {m.decisions} "
44
+ f"(completed {m.actions_completed}, rejected {m.rejections}, "
45
+ f"failed {m.execution_failures}, provider errors {m.provider_errors})"
46
+ )
47
+ print(f"model: {m.provider_calls} calls, {m.model_latency_s:.1f} s, tokens {tokens}")
48
+ print(f"sim time: {m.sim_time_s:.2f} s (wall {m.wall_time_s:.1f} s)")
49
+ if result.error:
50
+ print(f"error: {result.error}")
51
+ if result.trace_dir is not None:
52
+ video = result.trace_dir / "video.mp4"
53
+ print(f"trace: {result.trace_dir}")
54
+ print(f"video: {video if video.exists() else 'not recorded'}")
55
+
56
+
57
+ def main(argv: list[str] | None = None) -> int:
58
+ parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
59
+ parser.add_argument(
60
+ "--model",
61
+ default=os.environ.get("NOMAD_OPENAI_MODEL"),
62
+ help="OpenAI model ID (default: $NOMAD_OPENAI_MODEL)",
63
+ )
64
+ parser.add_argument("--seed", type=int, default=0)
65
+ parser.add_argument("--max-decisions", type=int, default=30)
66
+ parser.add_argument("--max-wall-time", type=float, default=600.0, help="seconds")
67
+ parser.add_argument("--reasoning-effort", default=None, help="e.g. low, medium, high")
68
+ parser.add_argument("--history-window", type=int, default=8, help="0 disables feedback")
69
+ parser.add_argument("--trace-root", type=Path, default=Path("runs"))
70
+ parser.add_argument("--no-video", action="store_true", help="skip video.mp4")
71
+ args = parser.parse_args(argv)
72
+ if not args.model:
73
+ parser.error("pass --model or set NOMAD_OPENAI_MODEL")
74
+
75
+ config = RunConfig(
76
+ seed=args.seed,
77
+ observation_mode=ObservationMode.RGB_PROPRIO,
78
+ max_decisions=args.max_decisions,
79
+ max_provider_calls=args.max_decisions,
80
+ max_wall_time_s=args.max_wall_time,
81
+ history_window=args.history_window,
82
+ trace_root=args.trace_root,
83
+ save_video=not args.no_video,
84
+ )
85
+ try:
86
+ model = OpenAI(
87
+ model=args.model,
88
+ reasoning_effort=args.reasoning_effort,
89
+ max_calls=args.max_decisions,
90
+ )
91
+ env = Robosuite(task="Lift", robot="Panda")
92
+ except (MissingDependencyError, ModelError) as exc:
93
+ print(f"error: {exc}", file=sys.stderr)
94
+ return 2
95
+
96
+ with model, env:
97
+ result = Agent(model, env, config).run(GOAL)
98
+
99
+ print_summary(result)
100
+ return 0 if result.status is RunStatus.SUCCESS else 1
101
+
102
+
103
+ if __name__ == "__main__":
104
+ raise SystemExit(main())
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
5
5
 
6
6
  [project]
7
7
  name = "nomad-harness"
8
- version = "0.1.0.dev0"
8
+ version = "0.1.0.dev1"
9
9
  description = "Action runtime connecting frontier models to simulated and physical embodiments."
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.10"
@@ -39,12 +39,15 @@ dev = [
39
39
  "mypy>=1.10",
40
40
  "jsonschema>=4.18",
41
41
  "types-PyYAML",
42
+ # The OpenAI adapter is type-checked and tested offline in CI.
43
+ "openai>=3.24,<4",
42
44
  ]
43
45
  # Integration extras. Version pins are set when each adapter is validated
44
46
  # against a concrete release. robosuite 1.5.2 does not bound mujoco, and newer
45
47
  # mujoco releases break it, so both are pinned to the tested pair.
46
48
  robosuite = ["robosuite==1.5.2", "mujoco==3.3.7", "numpy", "imageio[ffmpeg]"]
47
- openai = ["openai"]
49
+ # Request and response handling measured against openai 3.24.0 (Responses API).
50
+ openai = ["openai>=3.24,<4"]
48
51
  anthropic = ["anthropic"]
49
52
  gemini = ["google-genai"]
50
53
 
@@ -56,6 +59,7 @@ testpaths = ["tests"]
56
59
  addopts = "-ra --strict-markers"
57
60
  markers = [
58
61
  "robosuite: needs robosuite and mujoco installed (pip install -e '.[robosuite]')",
62
+ "openai_live: calls the live OpenAI API; opt in with NOMAD_OPENAI_LIVE=1",
59
63
  ]
60
64
 
61
65
  [tool.ruff]
@@ -29,7 +29,7 @@ from nomad.core import (
29
29
  )
30
30
  from nomad.runtime import Runtime
31
31
 
32
- __version__ = "0.1.0.dev0"
32
+ __version__ = "0.1.0.dev1"
33
33
 
34
34
  __all__ = [
35
35
  "Action",
@@ -22,14 +22,54 @@ Conventions:
22
22
  - set_gripper: closure 0 is fully open, 1 is fully closed. It is a setting, not a force.
23
23
  - duration_s is the maximum execution time; motion that does not converge reports a timeout.
24
24
 
25
+ Limits:
26
+ {limits}
27
+
25
28
  Rules:
26
- - Respond with exactly one action per decision, matching the provided schema.
29
+ - Respond with exactly one action per decision, using the provided action tools or schema.
27
30
  - Requests outside the limits are rejected, not clipped; read the feedback and adjust.
28
- - Set request_finish to true when you believe the task is complete. An independent check
29
- decides success; requesting finish does not make the task succeed.
31
+ - When you believe the task is complete, request finish (the finish tool, or request_finish
32
+ set to true). An independent check decides success; requesting finish does not make the
33
+ task succeed.
30
34
  """
31
35
 
32
36
 
37
+ def describe_limits(manifest: EmbodimentManifest) -> str:
38
+ """The manifest's action limits and workspace as instruction lines.
39
+
40
+ JSON Schema cannot express norm or workspace limits, so the model would
41
+ otherwise only learn them from rejections.
42
+ """
43
+ lines: list[str] = []
44
+ actions = manifest.actions
45
+ if actions.move_ee is not None:
46
+ spec = actions.move_ee
47
+ lines.append(
48
+ f"- move_ee: frame in {list(spec.frames)}; "
49
+ f"|translation_m| <= {spec.max_translation_norm_m:g} m; "
50
+ f"|rotation_vector_rad| <= {spec.max_rotation_norm_rad:g} rad; "
51
+ f"duration_s <= {spec.max_duration_s:g} s."
52
+ )
53
+ if actions.set_gripper is not None:
54
+ grip = actions.set_gripper
55
+ low, high = grip.range
56
+ lines.append(
57
+ f"- set_gripper: closure in [{low:g}, {high:g}]; "
58
+ f"duration_s <= {grip.max_duration_s:g} s."
59
+ )
60
+ if manifest.workspace is not None:
61
+ ws = manifest.workspace
62
+ bounds = ", ".join(
63
+ f"{axis} [{lo:g}, {hi:g}]"
64
+ for axis, lo, hi in zip("xyz", ws.min_m, ws.max_m, strict=True)
65
+ )
66
+ lines.append(
67
+ f"- The end-effector target (current position + translation_m) must stay inside "
68
+ f"the '{ws.frame}' workspace: {bounds} m."
69
+ )
70
+ return "\n".join(lines)
71
+
72
+
33
73
  class ContextBuilder:
34
74
  """Turns observations and feedback into :class:`AgentContext` records.
35
75
 
@@ -61,6 +101,7 @@ class ContextBuilder:
61
101
  robot=manifest.robot,
62
102
  task=manifest.task,
63
103
  pose_frame=manifest.pose_frame,
104
+ limits=describe_limits(manifest),
64
105
  )
65
106
 
66
107
  def record(self, entry: HistoryEntry) -> None:
@@ -60,6 +60,11 @@ class ModelResponse(Record):
60
60
 
61
61
  ``raw_proposal`` is untrusted: the runtime parses and validates it. Adapters
62
62
  must not pre-validate it into a :class:`~nomad.core.actions.Proposal`.
63
+
64
+ ``provider_details`` is optional JSON-compatible metadata for the trace,
65
+ e.g. the provider's response ID, finish reason, retry count, and a summary
66
+ of the request actually sent. Images are referenced by observation ID and
67
+ camera, never embedded, and credentials must never appear.
63
68
  """
64
69
 
65
70
  status: ModelResponseStatus
@@ -69,6 +74,7 @@ class ModelResponse(Record):
69
74
  raw_proposal: dict[str, Any] | str | None = None
70
75
  usage: Usage | None = None
71
76
  error: str | None = None
77
+ provider_details: dict[str, Any] | None = None
72
78
 
73
79
  @model_validator(mode="after")
74
80
  def _payload_matches_status(self) -> ModelResponse:
@@ -55,6 +55,10 @@ class Observation(Record):
55
55
  ``privileged`` may carry simulator-only state (object poses, etc.) for
56
56
  baselines and debugging. The context builder strips it unless the run is
57
57
  explicitly in ``privileged_state`` mode.
58
+
59
+ ``frame_poses`` maps declared manifest frames to their pose in ``world`` at
60
+ capture time (``world_from_<frame>``). It relates world-frame camera poses
61
+ to the robot-state pose frame; frames whose pose is unknown are omitted.
58
62
  """
59
63
 
60
64
  id: Annotated[str, Field(min_length=1)]
@@ -63,4 +67,5 @@ class Observation(Record):
63
67
  capture_wall_s: FiniteFloat
64
68
  frames: dict[str, CameraFrame]
65
69
  robot_state: RobotState
70
+ frame_poses: dict[str, Pose] = Field(default_factory=dict)
66
71
  privileged: dict[str, Any] | None = None
@@ -8,7 +8,8 @@ from typing import NamedTuple
8
8
 
9
9
  from nomad.core.errors import MissingDependencyError
10
10
 
11
- INSTALL_HINT = "pip install -e '.[robosuite]'"
11
+ # The installed-package form; from a source checkout, `pip install -e '.[robosuite]'` also works.
12
+ INSTALL_HINT = "pip install 'nomad-harness[robosuite]'"
12
13
 
13
14
 
14
15
  class Backend(NamedTuple):
@@ -154,6 +154,7 @@ class Robosuite:
154
154
  for stream, camera in self.config.cameras.items()
155
155
  },
156
156
  robot_state=self._robot_state(),
157
+ frame_poses={self._manifest.pose_frame: self._world_from_base()},
157
158
  privileged=self._privileged_state(),
158
159
  )
159
160
 
@@ -17,6 +17,7 @@ from pydantic import Field, PositiveInt
17
17
  from nomad.core.errors import ManifestError, NomadError
18
18
  from nomad.core.manifest import EmbodimentManifest, RgbStreamSpec
19
19
  from nomad.core.types import PositiveFloat, Record
20
+ from nomad.embodiments.robosuite._imports import INSTALL_HINT
20
21
 
21
22
  # Versions the adapter's conversions and defaults were measured against. robosuite
22
23
  # declares ``mujoco>=3.3.0`` without an upper bound, but newer mujoco releases break
@@ -148,6 +149,6 @@ def check_versions(installed: Mapping[str, str | None], *, allow_untested: bool)
148
149
  found = ", ".join(f"{n} {v}" for n, v in mismatched.items())
149
150
  raise UnsupportedVersionError(
150
151
  f"the robosuite adapter is validated with {wanted}, found {found}. Install the "
151
- "tested versions with `pip install -e '.[robosuite]'`, or set "
152
+ f"tested versions with `{INSTALL_HINT}`, or set "
152
153
  "RobosuiteConfig(allow_untested_versions=True) to try anyway."
153
154
  )
@@ -5,6 +5,7 @@ that a missing SDK only fails when that adapter is requested.
5
5
  """
6
6
 
7
7
  from nomad.models.fake import FakeFailure, FakeModel
8
+ from nomad.models.openai import OpenAI, OpenAIConfig
8
9
  from nomad.models.scripted import ScriptedLiftPolicy
9
10
 
10
- __all__ = ["FakeFailure", "FakeModel", "ScriptedLiftPolicy"]
11
+ __all__ = ["FakeFailure", "FakeModel", "OpenAI", "OpenAIConfig", "ScriptedLiftPolicy"]