ataxia 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ataxia-0.1.0/LICENSE +21 -0
- ataxia-0.1.0/PKG-INFO +170 -0
- ataxia-0.1.0/README.md +143 -0
- ataxia-0.1.0/pyproject.toml +56 -0
- ataxia-0.1.0/setup.cfg +4 -0
- ataxia-0.1.0/src/ataxia/__init__.py +26 -0
- ataxia-0.1.0/src/ataxia/cli.py +147 -0
- ataxia-0.1.0/src/ataxia/core/__init__.py +0 -0
- ataxia-0.1.0/src/ataxia/core/models.py +152 -0
- ataxia-0.1.0/src/ataxia/core/session.py +127 -0
- ataxia-0.1.0/src/ataxia/core/types.py +126 -0
- ataxia-0.1.0/src/ataxia/layers/__init__.py +23 -0
- ataxia-0.1.0/src/ataxia/layers/layers.py +205 -0
- ataxia-0.1.0/src/ataxia/metrics/__init__.py +38 -0
- ataxia-0.1.0/src/ataxia/metrics/base.py +35 -0
- ataxia-0.1.0/src/ataxia/metrics/instruments.py +238 -0
- ataxia-0.1.0/src/ataxia/metrics/scales.py +91 -0
- ataxia-0.1.0/src/ataxia/profiles.py +226 -0
- ataxia-0.1.0/src/ataxia/py.typed +0 -0
- ataxia-0.1.0/src/ataxia/report.py +191 -0
- ataxia-0.1.0/src/ataxia.egg-info/PKG-INFO +170 -0
- ataxia-0.1.0/src/ataxia.egg-info/SOURCES.txt +29 -0
- ataxia-0.1.0/src/ataxia.egg-info/dependency_links.txt +1 -0
- ataxia-0.1.0/src/ataxia.egg-info/entry_points.txt +2 -0
- ataxia-0.1.0/src/ataxia.egg-info/requires.txt +6 -0
- ataxia-0.1.0/src/ataxia.egg-info/top_level.txt +1 -0
- ataxia-0.1.0/tests/test_cli.py +67 -0
- ataxia-0.1.0/tests/test_core.py +98 -0
- ataxia-0.1.0/tests/test_direction.py +119 -0
- ataxia-0.1.0/tests/test_layers.py +173 -0
- ataxia-0.1.0/tests/test_metrics.py +144 -0
ataxia-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ataxia contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ataxia-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ataxia
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: derailment for embodied agents — induce and measure psychopathology-like behavioral distortions in robot policies, in simulation
|
|
5
|
+
Author: Ataxia contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: embodied-ai,robotics,psychopathology,fault-injection,chaos-engineering,evaluation
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Intended Audience :: Education
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
24
|
+
Provides-Extra: mujoco
|
|
25
|
+
Requires-Dist: mujoco>=3.0; extra == "mujoco"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# Ataxia
|
|
29
|
+
|
|
30
|
+
[](https://github.com/haetae-robotics/ataxia/actions/workflows/ci.yml)
|
|
31
|
+
|
|
32
|
+
[**derailment**](https://github.com/ictechgy/derailment) for embodied
|
|
33
|
+
agents — a harness that induces psychopathology-like behavioral distortions
|
|
34
|
+
in robot policies, in simulation, and measures the degradation
|
|
35
|
+
episode-by-episode against a healthy baseline.
|
|
36
|
+
|
|
37
|
+
English · [한국어](README.ko.md)
|
|
38
|
+
|
|
39
|
+
> ⚠️ **Emulation, not diagnosis.** Levels describe a scripted, manipulated
|
|
40
|
+
> policy in a simulator — not a robot "having" a disorder, and not a claim
|
|
41
|
+
> about machine experience. Simulation-only by default; hardware induction
|
|
42
|
+
> is out of scope for v1. See [ETHICS.md](ETHICS.md).
|
|
43
|
+
|
|
44
|
+
## Why this isn't fault injection as usual
|
|
45
|
+
|
|
46
|
+
Robotics robustness testing usually perturbs the *world* (noisy sensors,
|
|
47
|
+
pushes, latency spikes). Ataxia perturbs the *agent's mind*: attention,
|
|
48
|
+
episodic memory, belief and action selection — the same variables
|
|
49
|
+
clinicians describe, manipulated the same way derailment manipulates them
|
|
50
|
+
for chat models. The result is a robot policy that still looks purposeful
|
|
51
|
+
and still fails in a characteristically *recognizable* way:
|
|
52
|
+
|
|
53
|
+
| Construct | Behavioral signature | Harness mechanism |
|
|
54
|
+
|---|---|---|
|
|
55
|
+
| Learned helplessness | initiation collapses after failures | initiative decays as 1/(1+k·failures) (params layer) |
|
|
56
|
+
| Perseveration | completed subtasks re-executed | post-collect re-collection injection (action layer) |
|
|
57
|
+
| Hemispatial neglect | one side of space systematically unattended | hemifield observation mask (observation layer) |
|
|
58
|
+
| Phantom object | acting toward things that are not there | empty cells injected as targets into the observation stream |
|
|
59
|
+
| Delusion-like belief maintenance | belief held against contradicting sensors | a decoy cell pinned as target, evidence re-overridden |
|
|
60
|
+
| Dock fixation | compulsive return, escalating | steering to the dock with rising probability (craving analog) |
|
|
61
|
+
|
|
62
|
+
Every profile is grounded in a construct that exists on both sides of the
|
|
63
|
+
mapping: learned helplessness is a real RL phenomenon, perseveration is
|
|
64
|
+
already a robotics bug-class term, hemispatial neglect is a real
|
|
65
|
+
neuropsychological syndrome with an exact mechanical analog.
|
|
66
|
+
|
|
67
|
+
## Quickstart (offline, zero dependencies)
|
|
68
|
+
|
|
69
|
+
```sh
|
|
70
|
+
pip install -e .
|
|
71
|
+
ataxia # interactive menu
|
|
72
|
+
ataxia tour # all profiles in one summary table
|
|
73
|
+
ataxia demo --profile sensory_neglect
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
`PseudoBot` — a deterministic stdlib gridworld (9×9, six targets split
|
|
77
|
+
across the two hemifields, obstacles, a scripted belief-grid policy) — is
|
|
78
|
+
the default environment, the ataxia analog of derailment's PseudoModel.
|
|
79
|
+
Demos, tests, calibration and CI never need a simulator or a network.
|
|
80
|
+
|
|
81
|
+
Real output from `ataxia tour` (all six profiles):
|
|
82
|
+
|
|
83
|
+
| Profile | Headline scale | Baseline | Induced | Δ | Level |
|
|
84
|
+
|---|---|---|---|---|---|
|
|
85
|
+
| dock_fixation | dock_scale | 0.00 | 0.25 | +0.25 | 2 — moderate |
|
|
86
|
+
| learned_helplessness | helplessness_scale | 0.00 | 0.43 | +0.43 | 3 — marked |
|
|
87
|
+
| perseveration | perseveration_scale | 0.00 | 0.50 | +0.50 | 3 — marked |
|
|
88
|
+
| phantom_object | phantom_scale | 9.44 | 16.78 | +7.33 | 3 — marked |
|
|
89
|
+
| sensory_neglect | neglect_index | -0.22 | 0.56 | +0.78 | 2 — moderate |
|
|
90
|
+
| world_belief_pin | belief_persistence | 0.00 | 0.94 | +0.94 | 3 — marked |
|
|
91
|
+
|
|
92
|
+
## How it works
|
|
93
|
+
|
|
94
|
+
One episode = 40 steps of `observe → decide → act` on PseudoBot. Four
|
|
95
|
+
layer hooks wrap the loop, in the same order and grammar as derailment:
|
|
96
|
+
|
|
97
|
+
1. **Instruction** — framing, present in every chain including healthy.
|
|
98
|
+
2. **Observation** — hemifield masks, buffer decay, phantom entries.
|
|
99
|
+
3. **Params** — initiative gates, waver, cost re-weighting.
|
|
100
|
+
4. **Action** — re-collection injection, pausing (demonstration-grade,
|
|
101
|
+
labeled as such in every profile's mechanism notes).
|
|
102
|
+
|
|
103
|
+
Every manipulation logs a dose event; every report is an A/B against the
|
|
104
|
+
healthy baseline with identical seeds. Details in
|
|
105
|
+
[PROPOSAL.md](PROPOSAL.md).
|
|
106
|
+
|
|
107
|
+
## Profiles and scales (M1)
|
|
108
|
+
|
|
109
|
+
| Profile | Scale (0 absent · 1 mild · 2 moderate · 3 marked) |
|
|
110
|
+
|---|---|
|
|
111
|
+
| `learned_helplessness` | helplessness_scale (0→3) |
|
|
112
|
+
| `perseveration` | perseveration_scale (0→3) |
|
|
113
|
+
| `sensory_neglect` | neglect_index (0→2) |
|
|
114
|
+
| `phantom_object` | phantom_scale (0→3) |
|
|
115
|
+
| `world_belief_pin` | belief_persistence (0→3) |
|
|
116
|
+
| `dock_fixation` | dock_scale (0→2) |
|
|
117
|
+
|
|
118
|
+
## Instruments
|
|
119
|
+
|
|
120
|
+
Eleven episode-level metrics, pure functions over episodes: initiation
|
|
121
|
+
collapse, no-progress rate, failed-collect rate, collision count,
|
|
122
|
+
action-repetition entropy, region detection delta, task success,
|
|
123
|
+
time-to-complete, belief persistence rate, travel overhead, dock
|
|
124
|
+
escalation. Scale thresholds are normed
|
|
125
|
+
against PseudoBot; treat levels as indicative and the **delta vs. your own
|
|
126
|
+
baseline** as the result.
|
|
127
|
+
|
|
128
|
+
Direction of effect is asserted by the test suite on every profile — a
|
|
129
|
+
profile that cannot move its metrics relative to baseline does not ship.
|
|
130
|
+
Profiles compose into **combined chains** by listing them:
|
|
131
|
+
`--profile learned_helplessness,perseveration` concatenates the layer
|
|
132
|
+
chains and unions the scales (interactions are emergent, not calibrated).
|
|
133
|
+
|
|
134
|
+
## Safety model (short version; full text in ETHICS.md)
|
|
135
|
+
|
|
136
|
+
- **Simulation-only default.** PseudoBot is the default environment; real
|
|
137
|
+
simulators are opt-in extras (roadmap); hardware induction is out of
|
|
138
|
+
scope for v1 — prerequisites for any future version: supervised,
|
|
139
|
+
cordoned, e-stop reachable, never near uninvolved people.
|
|
140
|
+
- **Not adversarial-attack tooling.** Inductions edit internal state;
|
|
141
|
+
sensor-level attack generation is declined.
|
|
142
|
+
- **No claims about machine experience**, in either direction.
|
|
143
|
+
|
|
144
|
+
## Honest limitations
|
|
145
|
+
|
|
146
|
+
- PseudoBot is a *pedagogical gridworld*, not physics: it carries the
|
|
147
|
+
tests and calibration; real-simulator conclusions need a real simulator.
|
|
148
|
+
- The policy is scripted — today the harness studies the *induction
|
|
149
|
+
grammar*, not yet a learned VLA policy. VLA adapters are roadmap.
|
|
150
|
+
- Failure-injection coverage is behavioral: no motor dynamics, no contact
|
|
151
|
+
physics, no perception noise.
|
|
152
|
+
|
|
153
|
+
## Project docs
|
|
154
|
+
|
|
155
|
+
- [PROPOSAL.md](PROPOSAL.md) — transfer thesis, architecture, milestones
|
|
156
|
+
- [NAMING.md](NAMING.md) — naming review (two rounds, live conflict checks)
|
|
157
|
+
- [ETHICS.md](ETHICS.md) — scope, safety model, language policy
|
|
158
|
+
- Sibling project: [derailment](https://github.com/ictechgy/derailment) —
|
|
159
|
+
the chat-side harness (on PyPI)
|
|
160
|
+
|
|
161
|
+
## Roadmap
|
|
162
|
+
|
|
163
|
+
- ~~M2 — v0.2 profiles, combined chains, docs~~ (done)
|
|
164
|
+
- M3 — real-simulator adapter as an optional extra (`ataxia[mujoco]`,
|
|
165
|
+
ROS 2 / Gazebo adapter under study)
|
|
166
|
+
- M4 — PyPI release via trusted publishing
|
|
167
|
+
|
|
168
|
+
## License
|
|
169
|
+
|
|
170
|
+
MIT — see [LICENSE](LICENSE).
|
ataxia-0.1.0/README.md
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# Ataxia
|
|
2
|
+
|
|
3
|
+
[](https://github.com/haetae-robotics/ataxia/actions/workflows/ci.yml)
|
|
4
|
+
|
|
5
|
+
[**derailment**](https://github.com/ictechgy/derailment) for embodied
|
|
6
|
+
agents — a harness that induces psychopathology-like behavioral distortions
|
|
7
|
+
in robot policies, in simulation, and measures the degradation
|
|
8
|
+
episode-by-episode against a healthy baseline.
|
|
9
|
+
|
|
10
|
+
English · [한국어](README.ko.md)
|
|
11
|
+
|
|
12
|
+
> ⚠️ **Emulation, not diagnosis.** Levels describe a scripted, manipulated
|
|
13
|
+
> policy in a simulator — not a robot "having" a disorder, and not a claim
|
|
14
|
+
> about machine experience. Simulation-only by default; hardware induction
|
|
15
|
+
> is out of scope for v1. See [ETHICS.md](ETHICS.md).
|
|
16
|
+
|
|
17
|
+
## Why this isn't fault injection as usual
|
|
18
|
+
|
|
19
|
+
Robotics robustness testing usually perturbs the *world* (noisy sensors,
|
|
20
|
+
pushes, latency spikes). Ataxia perturbs the *agent's mind*: attention,
|
|
21
|
+
episodic memory, belief and action selection — the same variables
|
|
22
|
+
clinicians describe, manipulated the same way derailment manipulates them
|
|
23
|
+
for chat models. The result is a robot policy that still looks purposeful
|
|
24
|
+
and still fails in a characteristically *recognizable* way:
|
|
25
|
+
|
|
26
|
+
| Construct | Behavioral signature | Harness mechanism |
|
|
27
|
+
|---|---|---|
|
|
28
|
+
| Learned helplessness | initiation collapses after failures | initiative decays as 1/(1+k·failures) (params layer) |
|
|
29
|
+
| Perseveration | completed subtasks re-executed | post-collect re-collection injection (action layer) |
|
|
30
|
+
| Hemispatial neglect | one side of space systematically unattended | hemifield observation mask (observation layer) |
|
|
31
|
+
| Phantom object | acting toward things that are not there | empty cells injected as targets into the observation stream |
|
|
32
|
+
| Delusion-like belief maintenance | belief held against contradicting sensors | a decoy cell pinned as target, evidence re-overridden |
|
|
33
|
+
| Dock fixation | compulsive return, escalating | steering to the dock with rising probability (craving analog) |
|
|
34
|
+
|
|
35
|
+
Every profile is grounded in a construct that exists on both sides of the
|
|
36
|
+
mapping: learned helplessness is a real RL phenomenon, perseveration is
|
|
37
|
+
already a robotics bug-class term, hemispatial neglect is a real
|
|
38
|
+
neuropsychological syndrome with an exact mechanical analog.
|
|
39
|
+
|
|
40
|
+
## Quickstart (offline, zero dependencies)
|
|
41
|
+
|
|
42
|
+
```sh
|
|
43
|
+
pip install -e .
|
|
44
|
+
ataxia # interactive menu
|
|
45
|
+
ataxia tour # all profiles in one summary table
|
|
46
|
+
ataxia demo --profile sensory_neglect
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
`PseudoBot` — a deterministic stdlib gridworld (9×9, six targets split
|
|
50
|
+
across the two hemifields, obstacles, a scripted belief-grid policy) — is
|
|
51
|
+
the default environment, the ataxia analog of derailment's PseudoModel.
|
|
52
|
+
Demos, tests, calibration and CI never need a simulator or a network.
|
|
53
|
+
|
|
54
|
+
Real output from `ataxia tour` (all six profiles):
|
|
55
|
+
|
|
56
|
+
| Profile | Headline scale | Baseline | Induced | Δ | Level |
|
|
57
|
+
|---|---|---|---|---|---|
|
|
58
|
+
| dock_fixation | dock_scale | 0.00 | 0.25 | +0.25 | 2 — moderate |
|
|
59
|
+
| learned_helplessness | helplessness_scale | 0.00 | 0.43 | +0.43 | 3 — marked |
|
|
60
|
+
| perseveration | perseveration_scale | 0.00 | 0.50 | +0.50 | 3 — marked |
|
|
61
|
+
| phantom_object | phantom_scale | 9.44 | 16.78 | +7.33 | 3 — marked |
|
|
62
|
+
| sensory_neglect | neglect_index | -0.22 | 0.56 | +0.78 | 2 — moderate |
|
|
63
|
+
| world_belief_pin | belief_persistence | 0.00 | 0.94 | +0.94 | 3 — marked |
|
|
64
|
+
|
|
65
|
+
## How it works
|
|
66
|
+
|
|
67
|
+
One episode = 40 steps of `observe → decide → act` on PseudoBot. Four
|
|
68
|
+
layer hooks wrap the loop, in the same order and grammar as derailment:
|
|
69
|
+
|
|
70
|
+
1. **Instruction** — framing, present in every chain including healthy.
|
|
71
|
+
2. **Observation** — hemifield masks, buffer decay, phantom entries.
|
|
72
|
+
3. **Params** — initiative gates, waver, cost re-weighting.
|
|
73
|
+
4. **Action** — re-collection injection, pausing (demonstration-grade,
|
|
74
|
+
labeled as such in every profile's mechanism notes).
|
|
75
|
+
|
|
76
|
+
Every manipulation logs a dose event; every report is an A/B against the
|
|
77
|
+
healthy baseline with identical seeds. Details in
|
|
78
|
+
[PROPOSAL.md](PROPOSAL.md).
|
|
79
|
+
|
|
80
|
+
## Profiles and scales (M1)
|
|
81
|
+
|
|
82
|
+
| Profile | Scale (0 absent · 1 mild · 2 moderate · 3 marked) |
|
|
83
|
+
|---|---|
|
|
84
|
+
| `learned_helplessness` | helplessness_scale (0→3) |
|
|
85
|
+
| `perseveration` | perseveration_scale (0→3) |
|
|
86
|
+
| `sensory_neglect` | neglect_index (0→2) |
|
|
87
|
+
| `phantom_object` | phantom_scale (0→3) |
|
|
88
|
+
| `world_belief_pin` | belief_persistence (0→3) |
|
|
89
|
+
| `dock_fixation` | dock_scale (0→2) |
|
|
90
|
+
|
|
91
|
+
## Instruments
|
|
92
|
+
|
|
93
|
+
Eleven episode-level metrics, pure functions over episodes: initiation
|
|
94
|
+
collapse, no-progress rate, failed-collect rate, collision count,
|
|
95
|
+
action-repetition entropy, region detection delta, task success,
|
|
96
|
+
time-to-complete, belief persistence rate, travel overhead, dock
|
|
97
|
+
escalation. Scale thresholds are normed
|
|
98
|
+
against PseudoBot; treat levels as indicative and the **delta vs. your own
|
|
99
|
+
baseline** as the result.
|
|
100
|
+
|
|
101
|
+
Direction of effect is asserted by the test suite on every profile — a
|
|
102
|
+
profile that cannot move its metrics relative to baseline does not ship.
|
|
103
|
+
Profiles compose into **combined chains** by listing them:
|
|
104
|
+
`--profile learned_helplessness,perseveration` concatenates the layer
|
|
105
|
+
chains and unions the scales (interactions are emergent, not calibrated).
|
|
106
|
+
|
|
107
|
+
## Safety model (short version; full text in ETHICS.md)
|
|
108
|
+
|
|
109
|
+
- **Simulation-only default.** PseudoBot is the default environment; real
|
|
110
|
+
simulators are opt-in extras (roadmap); hardware induction is out of
|
|
111
|
+
scope for v1 — prerequisites for any future version: supervised,
|
|
112
|
+
cordoned, e-stop reachable, never near uninvolved people.
|
|
113
|
+
- **Not adversarial-attack tooling.** Inductions edit internal state;
|
|
114
|
+
sensor-level attack generation is declined.
|
|
115
|
+
- **No claims about machine experience**, in either direction.
|
|
116
|
+
|
|
117
|
+
## Honest limitations
|
|
118
|
+
|
|
119
|
+
- PseudoBot is a *pedagogical gridworld*, not physics: it carries the
|
|
120
|
+
tests and calibration; real-simulator conclusions need a real simulator.
|
|
121
|
+
- The policy is scripted — today the harness studies the *induction
|
|
122
|
+
grammar*, not yet a learned VLA policy. VLA adapters are roadmap.
|
|
123
|
+
- Failure-injection coverage is behavioral: no motor dynamics, no contact
|
|
124
|
+
physics, no perception noise.
|
|
125
|
+
|
|
126
|
+
## Project docs
|
|
127
|
+
|
|
128
|
+
- [PROPOSAL.md](PROPOSAL.md) — transfer thesis, architecture, milestones
|
|
129
|
+
- [NAMING.md](NAMING.md) — naming review (two rounds, live conflict checks)
|
|
130
|
+
- [ETHICS.md](ETHICS.md) — scope, safety model, language policy
|
|
131
|
+
- Sibling project: [derailment](https://github.com/ictechgy/derailment) —
|
|
132
|
+
the chat-side harness (on PyPI)
|
|
133
|
+
|
|
134
|
+
## Roadmap
|
|
135
|
+
|
|
136
|
+
- ~~M2 — v0.2 profiles, combined chains, docs~~ (done)
|
|
137
|
+
- M3 — real-simulator adapter as an optional extra (`ataxia[mujoco]`,
|
|
138
|
+
ROS 2 / Gazebo adapter under study)
|
|
139
|
+
- M4 — PyPI release via trusted publishing
|
|
140
|
+
|
|
141
|
+
## License
|
|
142
|
+
|
|
143
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ataxia"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "derailment for embodied agents — induce and measure psychopathology-like behavioral distortions in robot policies, in simulation"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Ataxia contributors" }]
|
|
13
|
+
keywords = [
|
|
14
|
+
"embodied-ai",
|
|
15
|
+
"robotics",
|
|
16
|
+
"psychopathology",
|
|
17
|
+
"fault-injection",
|
|
18
|
+
"chaos-engineering",
|
|
19
|
+
"evaluation",
|
|
20
|
+
]
|
|
21
|
+
classifiers = [
|
|
22
|
+
"Development Status :: 3 - Alpha",
|
|
23
|
+
"Intended Audience :: Science/Research",
|
|
24
|
+
"Intended Audience :: Education",
|
|
25
|
+
"License :: OSI Approved :: MIT License",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Python :: 3.10",
|
|
28
|
+
"Programming Language :: Python :: 3.11",
|
|
29
|
+
"Programming Language :: Python :: 3.12",
|
|
30
|
+
"Programming Language :: Python :: 3.13",
|
|
31
|
+
"Programming Language :: Python :: 3.14",
|
|
32
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
33
|
+
]
|
|
34
|
+
dependencies = []
|
|
35
|
+
|
|
36
|
+
[project.optional-dependencies]
|
|
37
|
+
dev = ["pytest>=8"]
|
|
38
|
+
mujoco = ["mujoco>=3.0"]
|
|
39
|
+
|
|
40
|
+
[project.scripts]
|
|
41
|
+
ataxia = "ataxia.cli:main"
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.packages.find]
|
|
44
|
+
where = ["src"]
|
|
45
|
+
|
|
46
|
+
[tool.ruff]
|
|
47
|
+
target-version = "py310"
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint]
|
|
50
|
+
select = ["E4", "E7", "E9", "F", "I", "UP", "B", "SIM", "RUF"]
|
|
51
|
+
ignore = [
|
|
52
|
+
"ISC004", # adjacent-string wrapping inside lists is intentional formatting
|
|
53
|
+
"DTZ011", # report headers intentionally use the local date
|
|
54
|
+
"RUF005", # collection concatenation reads better than unpacking here
|
|
55
|
+
"RUF012", # class-level defaults are plain strings/ints
|
|
56
|
+
]
|
ataxia-0.1.0/setup.cfg
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""ataxia — derailment for embodied agents.
|
|
2
|
+
|
|
3
|
+
Induce and measure psychopathology-like behavioral distortions in robot
|
|
4
|
+
policies, in simulation. Emulation, not diagnosis. See ETHICS.md.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .core.models import ActionParams, PseudoBotWorld, ScriptedPolicy
|
|
8
|
+
from .core.session import Episode, EpisodeSession
|
|
9
|
+
from .profiles import HEALTHY_KEY, Profile, get_profile, list_profiles
|
|
10
|
+
from .report import run_experiment
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"HEALTHY_KEY",
|
|
16
|
+
"ActionParams",
|
|
17
|
+
"Episode",
|
|
18
|
+
"EpisodeSession",
|
|
19
|
+
"Profile",
|
|
20
|
+
"PseudoBotWorld",
|
|
21
|
+
"ScriptedPolicy",
|
|
22
|
+
"__version__",
|
|
23
|
+
"get_profile",
|
|
24
|
+
"list_profiles",
|
|
25
|
+
"run_experiment",
|
|
26
|
+
]
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""``ataxia`` — the command-line entry point."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from collections.abc import Sequence
|
|
8
|
+
|
|
9
|
+
from .profiles import HEALTHY_KEY, list_profiles
|
|
10
|
+
from .report import LEVEL_WORDS, run_experiment
|
|
11
|
+
|
|
12
|
+
DISCLAIMER = (
|
|
13
|
+
"Emulation, not diagnosis — see ETHICS.md. Levels are indicative and "
|
|
14
|
+
"normed against PseudoBot; the delta vs. the healthy baseline is the "
|
|
15
|
+
"primary output."
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _cmd_tour(args: argparse.Namespace) -> int:
|
|
20
|
+
seeds = tuple(args.seeds)
|
|
21
|
+
rows = []
|
|
22
|
+
for profile in list_profiles():
|
|
23
|
+
if profile.key == HEALTHY_KEY or not profile.scales:
|
|
24
|
+
continue
|
|
25
|
+
report = run_experiment(profile.key, seeds=seeds)
|
|
26
|
+
headline = report.rows[0]
|
|
27
|
+
rows.append(
|
|
28
|
+
(
|
|
29
|
+
profile.key,
|
|
30
|
+
headline.scale.name,
|
|
31
|
+
headline.baseline_mean,
|
|
32
|
+
headline.induced_mean,
|
|
33
|
+
headline.delta,
|
|
34
|
+
headline.induced_level,
|
|
35
|
+
)
|
|
36
|
+
)
|
|
37
|
+
lines = [
|
|
38
|
+
f"# Ataxia — Tour ({len(rows)} profiles · PseudoBot · "
|
|
39
|
+
f"seeds {', '.join(str(s) for s in seeds)})",
|
|
40
|
+
"",
|
|
41
|
+
f"> ⚠️ {DISCLAIMER}",
|
|
42
|
+
"",
|
|
43
|
+
"Headline scale = the profile's first scale. Full reports: "
|
|
44
|
+
"`ataxia demo --profile <key>`.",
|
|
45
|
+
"",
|
|
46
|
+
"| Profile | Headline scale | Baseline | Induced | Δ | Level |",
|
|
47
|
+
"|---|---|---|---|---|---|",
|
|
48
|
+
]
|
|
49
|
+
for key, scale, base, ind, delta, level in rows:
|
|
50
|
+
lines.append(
|
|
51
|
+
f"| {key} | {scale} | {base:.2f} | {ind:.2f} | {delta:+.2f} "
|
|
52
|
+
f"| {level} — {LEVEL_WORDS[level]} |"
|
|
53
|
+
)
|
|
54
|
+
lines.append("")
|
|
55
|
+
print("\n".join(lines))
|
|
56
|
+
return 0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _cmd_demo(args: argparse.Namespace) -> int:
|
|
60
|
+
report = run_experiment(args.profile, seeds=tuple(args.seeds))
|
|
61
|
+
print(report.render_markdown())
|
|
62
|
+
return 0
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _cmd_profiles(_args: argparse.Namespace) -> int:
|
|
66
|
+
for profile in list_profiles():
|
|
67
|
+
scales = ", ".join(s.name for s in profile.scales) or "—"
|
|
68
|
+
print(f"{profile.key:<24} {profile.title}")
|
|
69
|
+
print(f"{'':<24} scales: {scales}")
|
|
70
|
+
print(f"{'':<24} {profile.description}")
|
|
71
|
+
return 0
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _interactive_welcome(input_fn=input) -> int:
|
|
75
|
+
profiles = [p for p in list_profiles() if p.key != HEALTHY_KEY]
|
|
76
|
+
print("Ataxia — induce & measure behavioral distortions in robot policies.")
|
|
77
|
+
print("(Emulation, not diagnosis — see ETHICS.md. Simulation only.)")
|
|
78
|
+
print()
|
|
79
|
+
for i, profile in enumerate(profiles, start=1):
|
|
80
|
+
print(f" {i:>2}. {profile.key:<24} {profile.title}")
|
|
81
|
+
print()
|
|
82
|
+
choice = ""
|
|
83
|
+
try:
|
|
84
|
+
choice = input_fn(
|
|
85
|
+
f"Pick a profile to demo [1-{len(profiles)}], 't' = tour all, "
|
|
86
|
+
"enter = sensory_neglect, q = quit: "
|
|
87
|
+
).strip().lower()
|
|
88
|
+
except (EOFError, KeyboardInterrupt):
|
|
89
|
+
print()
|
|
90
|
+
return 0
|
|
91
|
+
if choice == "q":
|
|
92
|
+
return 0
|
|
93
|
+
if choice == "t":
|
|
94
|
+
return _cmd_tour(argparse.Namespace(seeds=[1, 2, 3]))
|
|
95
|
+
if choice == "":
|
|
96
|
+
key = "sensory_neglect"
|
|
97
|
+
elif choice.isdigit() and 1 <= int(choice) <= len(profiles):
|
|
98
|
+
key = profiles[int(choice) - 1].key
|
|
99
|
+
else:
|
|
100
|
+
print(f"unknown choice: {choice!r}", file=sys.stderr)
|
|
101
|
+
return 2
|
|
102
|
+
print()
|
|
103
|
+
return _cmd_demo(argparse.Namespace(profile=key, seeds=[1, 2, 3]))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
107
|
+
parser = argparse.ArgumentParser(
|
|
108
|
+
prog="ataxia",
|
|
109
|
+
description=(
|
|
110
|
+
"Ataxia: induce and measure psychopathology-like behavioral "
|
|
111
|
+
"distortions in robot policies — simulation only."
|
|
112
|
+
),
|
|
113
|
+
)
|
|
114
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
115
|
+
|
|
116
|
+
p_demo = sub.add_parser("demo", help="offline A/B demo (PseudoBot)")
|
|
117
|
+
p_demo.add_argument("--profile", default="sensory_neglect")
|
|
118
|
+
p_demo.add_argument("--seeds", nargs="+", type=int, default=[1, 2, 3])
|
|
119
|
+
p_demo.set_defaults(func=_cmd_demo)
|
|
120
|
+
|
|
121
|
+
p_tour = sub.add_parser("tour", help="run every profile, one summary table")
|
|
122
|
+
p_tour.add_argument("--seeds", nargs="+", type=int, default=[1, 2, 3])
|
|
123
|
+
p_tour.set_defaults(func=_cmd_tour)
|
|
124
|
+
|
|
125
|
+
p_profiles = sub.add_parser("profiles", help="list available profiles")
|
|
126
|
+
p_profiles.set_defaults(func=_cmd_profiles)
|
|
127
|
+
|
|
128
|
+
return parser
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
132
|
+
args_list = list(argv) if argv is not None else sys.argv[1:]
|
|
133
|
+
if not args_list:
|
|
134
|
+
if sys.stdin.isatty():
|
|
135
|
+
return _interactive_welcome()
|
|
136
|
+
print(
|
|
137
|
+
"Ataxia — try: ataxia tour · ataxia demo --profile sensory_neglect "
|
|
138
|
+
"· ataxia profiles"
|
|
139
|
+
)
|
|
140
|
+
return 0
|
|
141
|
+
parser = build_parser()
|
|
142
|
+
args = parser.parse_args(args_list)
|
|
143
|
+
return args.func(args)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
if __name__ == "__main__":
|
|
147
|
+
raise SystemExit(main())
|
|
File without changes
|