edgeengine-aware 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgeengine_aware/__init__.py +93 -0
- edgeengine_aware/actions.py +194 -0
- edgeengine_aware/agriculture.py +191 -0
- edgeengine_aware/application.py +222 -0
- edgeengine_aware/communication.py +99 -0
- edgeengine_aware/config.py +942 -0
- edgeengine_aware/deployment.py +411 -0
- edgeengine_aware/domains.py +284 -0
- edgeengine_aware/energy.py +195 -0
- edgeengine_aware/env.py +441 -0
- edgeengine_aware/indoor.py +186 -0
- edgeengine_aware/industrial.py +192 -0
- edgeengine_aware/interfaces.py +171 -0
- edgeengine_aware/metrics.py +131 -0
- edgeengine_aware/observation.py +490 -0
- edgeengine_aware/policies.py +285 -0
- edgeengine_aware/process.py +136 -0
- edgeengine_aware/rendering.py +199 -0
- edgeengine_aware/reward.py +92 -0
- edgeengine_aware/rl.py +368 -0
- edgeengine_aware/scenarios.py +131 -0
- edgeengine_aware/sensing.py +45 -0
- edgeengine_aware/traces.py +636 -0
- edgeengine_aware-0.4.0.dist-info/METADATA +739 -0
- edgeengine_aware-0.4.0.dist-info/RECORD +28 -0
- edgeengine_aware-0.4.0.dist-info/WHEEL +5 -0
- edgeengine_aware-0.4.0.dist-info/licenses/LICENSE +21 -0
- edgeengine_aware-0.4.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Reward computation with explicit, separately reported components.
|
|
2
|
+
|
|
3
|
+
reward = application_utility
|
|
4
|
+
- sensing_cost - communication_cost
|
|
5
|
+
- staleness_penalty - battery_penalty - depletion_penalty
|
|
6
|
+
- rejection_penalty - waste_penalty
|
|
7
|
+
|
|
8
|
+
Scales (defaults, 7-day episode = 672 steps): the tracking utility is worth up
|
|
9
|
+
to 0.1 * criticality (1..3) per step, i.e. ~100-140 per episode for a policy
|
|
10
|
+
that keeps the application well informed; one high-quality sample + uplink
|
|
11
|
+
costs 0.12; the staleness penalty saturates at lambda_stale * w_priority
|
|
12
|
+
(0.05 / 0.10 / 0.20 per step) after tau_stale_s = 6 h - a policy that never
|
|
13
|
+
transmits loses ~120 per episode to it; the battery penalty grows
|
|
14
|
+
quadratically below ``safe_soc`` up to lambda_battery (0.5) per step at
|
|
15
|
+
SoC = 0, plus lambda_depletion (2.0) at every step in which the node actually
|
|
16
|
+
browns out. Energy penalties alone (~20 per episode for hourly high-quality
|
|
17
|
+
reporting in the standard mode: 168 x 0.12) do not dominate:
|
|
18
|
+
what makes energy binding is the battery penalty, which an always-on policy
|
|
19
|
+
pays to the tune of several hundred per episode.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from dataclasses import asdict, dataclass
|
|
25
|
+
|
|
26
|
+
from .config import RewardConfig
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class RewardComponents:
|
|
31
|
+
application_utility: float = 0.0
|
|
32
|
+
sensing_cost: float = 0.0
|
|
33
|
+
communication_cost: float = 0.0
|
|
34
|
+
staleness_penalty: float = 0.0
|
|
35
|
+
battery_penalty: float = 0.0
|
|
36
|
+
depletion_penalty: float = 0.0
|
|
37
|
+
rejection_penalty: float = 0.0
|
|
38
|
+
waste_penalty: float = 0.0
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def total(self) -> float:
|
|
42
|
+
return (
|
|
43
|
+
self.application_utility
|
|
44
|
+
- self.sensing_cost
|
|
45
|
+
- self.communication_cost
|
|
46
|
+
- self.staleness_penalty
|
|
47
|
+
- self.battery_penalty
|
|
48
|
+
- self.depletion_penalty
|
|
49
|
+
- self.rejection_penalty
|
|
50
|
+
- self.waste_penalty
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
def as_dict(self) -> dict[str, float]:
|
|
54
|
+
d = asdict(self)
|
|
55
|
+
d["total"] = self.total
|
|
56
|
+
return d
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class RewardCalculator:
|
|
60
|
+
def __init__(self, cfg: RewardConfig):
|
|
61
|
+
self.cfg = cfg
|
|
62
|
+
|
|
63
|
+
def compute(
|
|
64
|
+
self,
|
|
65
|
+
*,
|
|
66
|
+
utility: float,
|
|
67
|
+
sensing_energy_j: float,
|
|
68
|
+
communication_energy_j: float,
|
|
69
|
+
aoi_s: float,
|
|
70
|
+
priority: int,
|
|
71
|
+
soc_after: float,
|
|
72
|
+
depleted: bool,
|
|
73
|
+
n_rejected: int,
|
|
74
|
+
wasted_energy_j: float,
|
|
75
|
+
) -> RewardComponents:
|
|
76
|
+
c = self.cfg
|
|
77
|
+
w_prio = c.priority_weights[int(priority)]
|
|
78
|
+
staleness = c.lambda_stale * w_prio * min(aoi_s / c.tau_stale_s, 1.0)
|
|
79
|
+
if soc_after < c.safe_soc:
|
|
80
|
+
risk = ((c.safe_soc - soc_after) / c.safe_soc) ** 2
|
|
81
|
+
else:
|
|
82
|
+
risk = 0.0
|
|
83
|
+
return RewardComponents(
|
|
84
|
+
application_utility=utility,
|
|
85
|
+
sensing_cost=c.lambda_sense * sensing_energy_j / c.energy_ref_j,
|
|
86
|
+
communication_cost=c.lambda_tx * communication_energy_j / c.energy_ref_j,
|
|
87
|
+
staleness_penalty=staleness,
|
|
88
|
+
battery_penalty=c.lambda_battery * risk,
|
|
89
|
+
depletion_penalty=c.lambda_depletion if depleted else 0.0,
|
|
90
|
+
rejection_penalty=c.lambda_reject * n_rejected,
|
|
91
|
+
waste_penalty=c.lambda_waste * wasted_energy_j / c.energy_ref_j,
|
|
92
|
+
)
|
edgeengine_aware/rl.py
ADDED
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
"""Helpers for training and evaluating RL agents on EdgeEngine AWARE.
|
|
2
|
+
|
|
3
|
+
Nothing here depends on a specific RL library except :class:`SB3Policy`, which
|
|
4
|
+
only needs an object with a ``predict(obs, deterministic=True)`` method (the
|
|
5
|
+
Stable-Baselines3 convention). Import of Stable-Baselines3 itself is left to
|
|
6
|
+
the caller, so the core package stays dependency-light.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Callable, Iterable, Mapping
|
|
13
|
+
|
|
14
|
+
import gymnasium as gym
|
|
15
|
+
import numpy as np
|
|
16
|
+
from gymnasium import spaces
|
|
17
|
+
|
|
18
|
+
from .actions import flatten_action, n_flat_actions, unflatten_action
|
|
19
|
+
from .env import EdgeEngineAwareEnv
|
|
20
|
+
from .interfaces import Policy
|
|
21
|
+
from .observation import OBSERVATION_FIELDS
|
|
22
|
+
from .policies import run_episode
|
|
23
|
+
from .scenarios import get_scenario, scenario_names
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
# Wrappers
|
|
28
|
+
# ---------------------------------------------------------------------------
|
|
29
|
+
class FlatActionWrapper(gym.ActionWrapper):
|
|
30
|
+
"""Expose the ``MultiDiscrete([3, 1 + n_modes])`` action as ``Discrete(3 * (1 + n_modes))``.
|
|
31
|
+
|
|
32
|
+
``flat = sensing_level * (1 + n_modes) + transmit`` (see ``actions.py``).
|
|
33
|
+
Needed by value-based agents such as DQN; PPO/A2C work on the
|
|
34
|
+
MultiDiscrete space directly.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
def __init__(self, env: gym.Env):
|
|
38
|
+
super().__init__(env)
|
|
39
|
+
self.n_modes = int(env.action_space.nvec[1]) - 1
|
|
40
|
+
self.action_space = spaces.Discrete(n_flat_actions(self.n_modes))
|
|
41
|
+
|
|
42
|
+
def action(self, action):
|
|
43
|
+
return unflatten_action(int(action), self.n_modes)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class MixedScenarioEnv(EdgeEngineAwareEnv):
|
|
47
|
+
"""EdgeEngine AWARE environment that draws a *scenario* at every reset.
|
|
48
|
+
|
|
49
|
+
Training on a mixture of scenarios (plus domain randomisation inside each)
|
|
50
|
+
is the simplest way to obtain a policy that is robust to weather, storage
|
|
51
|
+
size, link quality and application behaviour, instead of one tuned to the
|
|
52
|
+
nominal configuration. The scenario of the current episode is reported in
|
|
53
|
+
``info["scenario"]``.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
def __init__(self, scenarios: Iterable[str], *, randomize: bool = True, render_mode: str | None = None):
|
|
57
|
+
self.scenario_names = list(scenarios)
|
|
58
|
+
if not self.scenario_names:
|
|
59
|
+
raise ValueError("at least one scenario name is required")
|
|
60
|
+
self._configs = {n: get_scenario(n, randomize=randomize) for n in self.scenario_names}
|
|
61
|
+
self.current_scenario = self.scenario_names[0]
|
|
62
|
+
super().__init__(self._configs[self.current_scenario], render_mode=render_mode)
|
|
63
|
+
|
|
64
|
+
def reset(self, *, seed: int | None = None, options: dict[str, Any] | None = None):
|
|
65
|
+
if seed is not None: # seed the scenario draw as well, for reproducibility
|
|
66
|
+
super().reset(seed=seed)
|
|
67
|
+
self.current_scenario = self.scenario_names[int(self.np_random.integers(len(self.scenario_names)))]
|
|
68
|
+
self.base_config = self._configs[self.current_scenario]
|
|
69
|
+
obs, info = super().reset(seed=None, options=options)
|
|
70
|
+
info["scenario"] = self.current_scenario
|
|
71
|
+
return obs, info
|
|
72
|
+
|
|
73
|
+
def step(self, action):
|
|
74
|
+
obs, reward, terminated, truncated, info = super().step(action)
|
|
75
|
+
info["scenario"] = self.current_scenario
|
|
76
|
+
return obs, reward, terminated, truncated, info
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def make_env(scenario: str | Iterable[str] = "default", *, randomize: bool = False, flat_actions: bool = False, seed: int | None = None, render_mode: str | None = None) -> gym.Env:
|
|
80
|
+
"""Build an environment for a named scenario, or for a mixture of scenarios
|
|
81
|
+
(``"mixed"`` = all of them, or an explicit list of names), optionally with
|
|
82
|
+
domain randomisation and flat actions. ``seed`` seeds the first reset."""
|
|
83
|
+
if isinstance(scenario, str) and (scenario == "mixed" or scenario.startswith("mixed:") or scenario == "all"):
|
|
84
|
+
# "mixed" = all agricultural scenarios (historical default), "mixed:<domain>" = one
|
|
85
|
+
# domain's scenarios, "all" = every scenario of every domain
|
|
86
|
+
scenario = scenario_names("all" if scenario == "all" else ("agriculture" if scenario == "mixed" else scenario.split(":", 1)[1]))
|
|
87
|
+
if isinstance(scenario, str):
|
|
88
|
+
env: gym.Env = EdgeEngineAwareEnv(get_scenario(scenario, randomize=randomize), render_mode=render_mode)
|
|
89
|
+
else:
|
|
90
|
+
env = MixedScenarioEnv(scenario, randomize=randomize, render_mode=render_mode)
|
|
91
|
+
if flat_actions:
|
|
92
|
+
env = FlatActionWrapper(env)
|
|
93
|
+
if seed is not None:
|
|
94
|
+
env.reset(seed=seed)
|
|
95
|
+
return env
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def make_env_fn(scenario: str | Iterable[str] = "default", *, randomize: bool = True, flat_actions: bool = False, seed: int = 0) -> Callable[[], gym.Env]:
|
|
99
|
+
"""Factory usable with SB3 ``DummyVecEnv`` / ``SubprocVecEnv``."""
|
|
100
|
+
|
|
101
|
+
def _init() -> gym.Env:
|
|
102
|
+
return make_env(scenario, randomize=randomize, flat_actions=flat_actions, seed=seed)
|
|
103
|
+
|
|
104
|
+
return _init
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
# Policy adapter
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
class SB3Policy:
|
|
111
|
+
"""Wrap a trained Stable-Baselines3 model as an EdgeEngine AWARE ``Policy``.
|
|
112
|
+
|
|
113
|
+
The adapter always returns the MultiDiscrete action array, undoing the flat
|
|
114
|
+
encoding when the model was trained with :class:`FlatActionWrapper`.
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
def __init__(self, model: Any, flat_actions: bool = False, deterministic: bool = True):
|
|
118
|
+
self.model = model
|
|
119
|
+
self.flat_actions = flat_actions
|
|
120
|
+
self.deterministic = deterministic
|
|
121
|
+
# number of radio modes, recovered from the model's action space
|
|
122
|
+
space = model.action_space
|
|
123
|
+
self.n_modes = (int(space.n) // 3 - 1) if flat_actions else int(space.nvec[1]) - 1
|
|
124
|
+
self.reset()
|
|
125
|
+
|
|
126
|
+
def reset(self) -> None:
|
|
127
|
+
self._state = None # hidden state of a recurrent policy (RecurrentPPO); None for feed-forward models
|
|
128
|
+
|
|
129
|
+
def act(self, observation) -> np.ndarray:
|
|
130
|
+
action, self._state = self.model.predict(np.asarray(observation, dtype=np.float32), state=self._state, deterministic=self.deterministic)
|
|
131
|
+
if self.flat_actions:
|
|
132
|
+
return unflatten_action(int(np.asarray(action).reshape(-1)[0]), self.n_modes)
|
|
133
|
+
return np.asarray(action, dtype=np.int64).reshape(-1)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# ---------------------------------------------------------------------------
|
|
138
|
+
# Evaluation protocol
|
|
139
|
+
# ---------------------------------------------------------------------------
|
|
140
|
+
@dataclass
|
|
141
|
+
class EvalRow:
|
|
142
|
+
policy: str
|
|
143
|
+
scenario: str
|
|
144
|
+
seed: int
|
|
145
|
+
reward: float
|
|
146
|
+
utility: float
|
|
147
|
+
harvested_j: float
|
|
148
|
+
consumed_j: float
|
|
149
|
+
n_sensing: int
|
|
150
|
+
n_high_quality: int
|
|
151
|
+
n_tx: int
|
|
152
|
+
n_delivered: int
|
|
153
|
+
min_soc: float
|
|
154
|
+
low_battery_frac: float
|
|
155
|
+
depletions: int
|
|
156
|
+
aoi_h: float
|
|
157
|
+
max_aoi_h: float
|
|
158
|
+
components: dict[str, float] = field(default_factory=dict)
|
|
159
|
+
|
|
160
|
+
def as_dict(self) -> dict[str, Any]:
|
|
161
|
+
d = {k: v for k, v in self.__dict__.items() if k != "components"}
|
|
162
|
+
d.update({f"c_{k}": v for k, v in self.components.items()})
|
|
163
|
+
return d
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def evaluate(
|
|
167
|
+
policies: dict[str, Callable[[], Policy]],
|
|
168
|
+
scenarios: Iterable[str] = ("default",),
|
|
169
|
+
seeds: Iterable[int] = range(1000, 1010),
|
|
170
|
+
*,
|
|
171
|
+
randomize: bool = False,
|
|
172
|
+
progress: Callable[[str], None] | None = None,
|
|
173
|
+
envs: Mapping[str, gym.Env] | None = None,
|
|
174
|
+
) -> list[EvalRow]:
|
|
175
|
+
"""Run every policy on every scenario and seed; return one row per episode.
|
|
176
|
+
|
|
177
|
+
Policies are given as factories so that stateful policies start fresh.
|
|
178
|
+
Seeds default to a held-out range (1000+) that training never touches.
|
|
179
|
+
``envs`` replaces the named scenarios by ready-made environments
|
|
180
|
+
(``{label: env}``), e.g. trace-driven ones; the label is reported as the
|
|
181
|
+
row's ``scenario``.
|
|
182
|
+
"""
|
|
183
|
+
rows: list[EvalRow] = []
|
|
184
|
+
seeds = list(seeds)
|
|
185
|
+
if envs is not None:
|
|
186
|
+
targets: list[tuple[str, gym.Env, bool]] = [(k, v, False) for k, v in envs.items()]
|
|
187
|
+
else:
|
|
188
|
+
targets = [(str(s), make_env(s, randomize=randomize), True) for s in scenarios]
|
|
189
|
+
for scenario, env, owned in targets:
|
|
190
|
+
for name, factory in policies.items():
|
|
191
|
+
if progress:
|
|
192
|
+
progress(f"{scenario:22s} {name}")
|
|
193
|
+
for seed in seeds:
|
|
194
|
+
res = run_episode(env, factory(), seed=seed)
|
|
195
|
+
m = env.unwrapped.metrics
|
|
196
|
+
rows.append(
|
|
197
|
+
EvalRow(
|
|
198
|
+
policy=name,
|
|
199
|
+
scenario=scenario,
|
|
200
|
+
seed=seed,
|
|
201
|
+
reward=res.total_reward,
|
|
202
|
+
utility=m.total_application_utility,
|
|
203
|
+
harvested_j=m.total_harvested_energy_j,
|
|
204
|
+
consumed_j=m.total_consumed_energy_j,
|
|
205
|
+
n_sensing=m.n_sensing,
|
|
206
|
+
n_high_quality=m.n_high_quality_sensing,
|
|
207
|
+
n_tx=m.n_transmissions,
|
|
208
|
+
n_delivered=m.n_successful_transmissions,
|
|
209
|
+
min_soc=m.min_battery_soc,
|
|
210
|
+
low_battery_frac=m.fraction_low_battery,
|
|
211
|
+
depletions=m.battery_depletion_events,
|
|
212
|
+
aoi_h=m.average_aoi_s / 3600.0,
|
|
213
|
+
max_aoi_h=m.max_aoi_s / 3600.0,
|
|
214
|
+
components=dict(m.reward_components),
|
|
215
|
+
)
|
|
216
|
+
)
|
|
217
|
+
if owned:
|
|
218
|
+
env.close()
|
|
219
|
+
return rows
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def summarize(rows: list[EvalRow], metric: str = "reward") -> dict[str, dict[str, tuple[float, float]]]:
|
|
223
|
+
"""``{scenario: {policy: (mean, std)}}`` for one metric."""
|
|
224
|
+
out: dict[str, dict[str, tuple[float, float]]] = {}
|
|
225
|
+
for r in rows:
|
|
226
|
+
out.setdefault(r.scenario, {}).setdefault(r.policy, []) # type: ignore[arg-type]
|
|
227
|
+
out[r.scenario][r.policy].append(getattr(r, metric)) # type: ignore[union-attr]
|
|
228
|
+
return {s: {p: (float(np.mean(v)), float(np.std(v))) for p, v in d.items()} for s, d in out.items()}
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
__all__ = ["FlatActionWrapper", "MixedScenarioEnv", "make_env", "make_env_fn", "SB3Policy", "EvalRow", "evaluate", "summarize", "flatten_action"]
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# ---------------------------------------------------------------------------
|
|
235
|
+
# Export of small MLP policies (SB3 -> plain numbers -> numpy runtime)
|
|
236
|
+
# ---------------------------------------------------------------------------
|
|
237
|
+
def export_sb3_mlp(model: Any) -> dict[str, Any]:
|
|
238
|
+
"""Extract the actor of an SB3 ``PPO``/``A2C`` (MultiDiscrete) or ``DQN``
|
|
239
|
+
(flat Discrete) MlpPolicy as nested lists, ready for ``PolicyBundle.model``.
|
|
240
|
+
|
|
241
|
+
Returned dict::
|
|
242
|
+
|
|
243
|
+
{"type": "mlp", "input_dim": 18,
|
|
244
|
+
"layers": [{"W": [[...]], "b": [...], "activation": "tanh"|"relu"|"linear"}, ...],
|
|
245
|
+
"output": "multidiscrete_logits" | "flat_q_values",
|
|
246
|
+
"output_split": [3, 4], # only for multidiscrete_logits
|
|
247
|
+
"n_modes": 3}
|
|
248
|
+
"""
|
|
249
|
+
import torch # local import: torch is only needed when exporting
|
|
250
|
+
|
|
251
|
+
policy = model.policy
|
|
252
|
+
layers: list[dict[str, Any]] = []
|
|
253
|
+
|
|
254
|
+
def add_sequential(seq, final_activation: str | None = None):
|
|
255
|
+
mods = [m for m in seq if not isinstance(m, torch.nn.Flatten)]
|
|
256
|
+
pending: dict | None = None
|
|
257
|
+
for m in mods:
|
|
258
|
+
if isinstance(m, torch.nn.Linear):
|
|
259
|
+
if pending is not None:
|
|
260
|
+
pending["activation"] = "linear"
|
|
261
|
+
layers.append(pending)
|
|
262
|
+
pending = {"W": m.weight.detach().cpu().numpy().tolist(), "b": m.bias.detach().cpu().numpy().tolist()}
|
|
263
|
+
elif isinstance(m, (torch.nn.Tanh, torch.nn.ReLU)):
|
|
264
|
+
assert pending is not None, "activation without a preceding Linear layer"
|
|
265
|
+
pending["activation"] = "tanh" if isinstance(m, torch.nn.Tanh) else "relu"
|
|
266
|
+
layers.append(pending)
|
|
267
|
+
pending = None
|
|
268
|
+
else:
|
|
269
|
+
raise TypeError(f"unsupported module in policy network: {type(m).__name__}")
|
|
270
|
+
if pending is not None:
|
|
271
|
+
pending["activation"] = final_activation or "linear"
|
|
272
|
+
layers.append(pending)
|
|
273
|
+
|
|
274
|
+
if hasattr(policy, "q_net"): # DQN
|
|
275
|
+
add_sequential(policy.q_net.q_net, final_activation="linear")
|
|
276
|
+
output, split = "flat_q_values", None
|
|
277
|
+
n_modes = int(model.action_space.n) // 3 - 1
|
|
278
|
+
else: # on-policy actor-critic
|
|
279
|
+
add_sequential(policy.mlp_extractor.policy_net)
|
|
280
|
+
add_sequential(torch.nn.Sequential(policy.action_net), final_activation="linear")
|
|
281
|
+
output = "multidiscrete_logits"
|
|
282
|
+
split = [int(n) for n in model.action_space.nvec]
|
|
283
|
+
n_modes = split[1] - 1
|
|
284
|
+
return {"type": "mlp", "input_dim": len(layers[0]["W"][0]), "layers": layers, "output": output, "output_split": split, "n_modes": n_modes}
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
class NumpyMLPPolicy:
|
|
288
|
+
"""Dependency-free runtime for an exported MLP (what a C port would do).
|
|
289
|
+
|
|
290
|
+
Runs the forward pass in float32 with numpy only and applies the argmax
|
|
291
|
+
decoding of the action encoding, so a bundle exported with
|
|
292
|
+
:func:`export_sb3_mlp` can be executed without torch or SB3 - and compared
|
|
293
|
+
action-by-action with the original model (see the RL notebook).
|
|
294
|
+
"""
|
|
295
|
+
|
|
296
|
+
def __init__(self, model: dict[str, Any]):
|
|
297
|
+
self.layers = [(np.asarray(l["W"], dtype=np.float32), np.asarray(l["b"], dtype=np.float32), l["activation"]) for l in model["layers"]]
|
|
298
|
+
self.output = model["output"]
|
|
299
|
+
self.split = model.get("output_split")
|
|
300
|
+
self.n_modes = int(model.get("n_modes", (self.split[1] - 1) if self.split else 3))
|
|
301
|
+
|
|
302
|
+
def reset(self) -> None:
|
|
303
|
+
pass
|
|
304
|
+
|
|
305
|
+
def forward(self, observation) -> np.ndarray:
|
|
306
|
+
h = np.asarray(observation, dtype=np.float32).reshape(-1)
|
|
307
|
+
for W, b, act in self.layers:
|
|
308
|
+
h = W @ h + b
|
|
309
|
+
if act == "tanh":
|
|
310
|
+
h = np.tanh(h)
|
|
311
|
+
elif act == "relu":
|
|
312
|
+
h = np.maximum(h, 0.0)
|
|
313
|
+
return h
|
|
314
|
+
|
|
315
|
+
def act(self, observation) -> np.ndarray:
|
|
316
|
+
out = self.forward(observation)
|
|
317
|
+
if self.output == "flat_q_values":
|
|
318
|
+
return unflatten_action(int(np.argmax(out)), self.n_modes)
|
|
319
|
+
n1 = self.split[0]
|
|
320
|
+
return np.array([int(np.argmax(out[:n1])), int(np.argmax(out[n1:]))], dtype=np.int64)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
__all__ += ["export_sb3_mlp", "NumpyMLPPolicy"]
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
# ---------------------------------------------------------------------------
|
|
327
|
+
# Memory: frame stacking (mirrors stable_baselines3 VecFrameStack for 1-D obs)
|
|
328
|
+
# ---------------------------------------------------------------------------
|
|
329
|
+
class FrameStacker:
|
|
330
|
+
"""Keep the last ``n_stack`` observations concatenated (oldest first).
|
|
331
|
+
|
|
332
|
+
Identical to SB3's ``VecFrameStack`` for 1-D observations: the buffer is
|
|
333
|
+
zero-filled at reset and shifted left at every step. On a microcontroller
|
|
334
|
+
this is a ring buffer of ``n_stack`` observation vectors.
|
|
335
|
+
"""
|
|
336
|
+
|
|
337
|
+
def __init__(self, n_stack: int, obs_dim: int):
|
|
338
|
+
self.n_stack, self.obs_dim = int(n_stack), int(obs_dim)
|
|
339
|
+
self.reset()
|
|
340
|
+
|
|
341
|
+
def reset(self) -> None:
|
|
342
|
+
self._buf = np.zeros(self.n_stack * self.obs_dim, dtype=np.float32)
|
|
343
|
+
|
|
344
|
+
def push(self, observation) -> np.ndarray:
|
|
345
|
+
o = np.asarray(observation, dtype=np.float32).reshape(-1)
|
|
346
|
+
self._buf = np.concatenate([self._buf[self.obs_dim:], o])
|
|
347
|
+
return self._buf.copy()
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
class StackedPolicy:
|
|
351
|
+
"""Wrap any ``Policy`` that expects a stacked observation (e.g. an
|
|
352
|
+
:class:`SB3Policy` of a model trained with ``VecFrameStack``, or a
|
|
353
|
+
:class:`NumpyMLPPolicy` exported from it) so that it can be used where a
|
|
354
|
+
plain observation is provided (``run_episode``, ``evaluate``, the firmware loop)."""
|
|
355
|
+
|
|
356
|
+
def __init__(self, inner: Policy, n_stack: int, obs_dim: int = len(OBSERVATION_FIELDS)):
|
|
357
|
+
self.inner = inner
|
|
358
|
+
self.stacker = FrameStacker(n_stack, obs_dim)
|
|
359
|
+
|
|
360
|
+
def reset(self) -> None:
|
|
361
|
+
self.stacker.reset()
|
|
362
|
+
self.inner.reset()
|
|
363
|
+
|
|
364
|
+
def act(self, observation) -> np.ndarray:
|
|
365
|
+
return self.inner.act(self.stacker.push(observation))
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
__all__ += ["FrameStacker", "StackedPolicy"]
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Named evaluation scenarios of the agriculture domain (and the lookup of
|
|
2
|
+
every domain's scenarios by qualified name).
|
|
3
|
+
|
|
4
|
+
A scenario is a function that returns a fully configured
|
|
5
|
+
:class:`EdgeEngineAwareConfig`. The set below is the *benchmark protocol* of the
|
|
6
|
+
project: policies are compared on every scenario, over a fixed set of seeds,
|
|
7
|
+
with the same metrics. ``default`` is the training distribution; the others
|
|
8
|
+
stress one aspect of the problem so that the value of an adaptive policy
|
|
9
|
+
becomes visible where a fixed duty cycle cannot cope. The scenarios of the
|
|
10
|
+
other domains live in ``domains.py`` and are addressed as ``"indoor_air:no_window"``
|
|
11
|
+
or ``"industrial:degrading"`` (``scenario_names("all")`` lists everything).
|
|
12
|
+
|
|
13
|
+
Scenario what is stressed what a good policy does
|
|
14
|
+
---------------------- ------------------------------------------ --------------------------------------------
|
|
15
|
+
default nothing in particular (mixed weather) report ~hourly, more near thresholds
|
|
16
|
+
cloudy_week harvesting ~55 % of default (30 J/day), slow down early, keep a reserve, use cheap checks
|
|
17
|
+
battery starts at 40 %
|
|
18
|
+
tiny_battery storage halved (150 J), small buffer smooth consumption, avoid bursts at night
|
|
19
|
+
lossy_link +6 dB path loss, slower fading choose the radio mode from the link estimate
|
|
20
|
+
drought fast drying, almost no rain, slow track the approach to the thresholds closely
|
|
21
|
+
irrigation
|
|
22
|
+
demanding_application frequent elevated/urgent campaigns follow the priority, save energy in between
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from typing import Callable
|
|
28
|
+
|
|
29
|
+
from .config import HOUR_S, EdgeEngineAwareConfig, default_config
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def default() -> EdgeEngineAwareConfig:
|
|
33
|
+
return default_config()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def cloudy_week() -> EdgeEngineAwareConfig:
|
|
37
|
+
cfg = default_config()
|
|
38
|
+
cfg.harvesting.clearness_mean = 0.35
|
|
39
|
+
cfg.harvesting.clearness_std = 0.15
|
|
40
|
+
cfg.harvesting.cloud_noise_std = 0.25
|
|
41
|
+
cfg.storage.initial_soc = 0.4
|
|
42
|
+
return cfg
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def tiny_battery() -> EdgeEngineAwareConfig:
|
|
46
|
+
cfg = default_config()
|
|
47
|
+
cfg.storage.capacity_j = 150.0
|
|
48
|
+
cfg.storage.initial_soc = 0.5
|
|
49
|
+
return cfg
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def lossy_link() -> EdgeEngineAwareConfig:
|
|
53
|
+
"""Node far from the gateway: +6 dB path loss and slower fading (AR(1)
|
|
54
|
+
coefficient 0.98 instead of 0.97, i.e. bad phases last longer). Mean
|
|
55
|
+
margins become -8 / -2 / +6 dB for fast / standard / robust; averaged over
|
|
56
|
+
the fading, the standard mode delivers ~40 % and the robust mode ~85 % of
|
|
57
|
+
the uplinks at twice the energy; the fast mode only works in favourable
|
|
58
|
+
fading phases."""
|
|
59
|
+
cfg = default_config()
|
|
60
|
+
cfg.communication.path_loss_mean_db += 6.0
|
|
61
|
+
cfg.communication.slow_fading_autocorr = 0.98
|
|
62
|
+
return cfg
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def drought() -> EdgeEngineAwareConfig:
|
|
66
|
+
cfg = default_config()
|
|
67
|
+
cfg.agriculture.et_rate_per_day = 0.12
|
|
68
|
+
cfg.agriculture.rain_events_per_day = 0.02
|
|
69
|
+
cfg.agriculture.irrigation_delay_mean_s = 18.0 * HOUR_S
|
|
70
|
+
cfg.agriculture.initial_moisture_range = (0.40, 0.55)
|
|
71
|
+
return cfg
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def demanding_application() -> EdgeEngineAwareConfig:
|
|
75
|
+
cfg = default_config()
|
|
76
|
+
cfg.application.request_rate_per_day = 1.5
|
|
77
|
+
cfg.application.request_duration_range_s = (2.0 * HOUR_S, 8.0 * HOUR_S)
|
|
78
|
+
cfg.application.request_urgent_fraction = 0.5
|
|
79
|
+
cfg.application.aoi_elevated_s = 4.0 * HOUR_S
|
|
80
|
+
return cfg
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
SCENARIOS: dict[str, Callable[[], EdgeEngineAwareConfig]] = {
|
|
84
|
+
"default": default,
|
|
85
|
+
"cloudy_week": cloudy_week,
|
|
86
|
+
"tiny_battery": tiny_battery,
|
|
87
|
+
"lossy_link": lossy_link,
|
|
88
|
+
"drought": drought,
|
|
89
|
+
"demanding_application": demanding_application,
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def split_scenario_name(name: str) -> tuple[str, str]:
|
|
94
|
+
"""``"indoor_air:no_window"`` -> ``("indoor_air", "no_window")``; a bare name is agricultural."""
|
|
95
|
+
if ":" in name:
|
|
96
|
+
domain, scen = name.split(":", 1)
|
|
97
|
+
return domain, scen
|
|
98
|
+
return "agriculture", name
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def scenario_names(domain: str = "agriculture") -> list[str]:
|
|
102
|
+
"""Qualified names (``domain:scenario``) of the benchmark scenarios of ``domain``,
|
|
103
|
+
or of every domain with ``domain="all"``. Agricultural names are returned bare
|
|
104
|
+
for backwards compatibility."""
|
|
105
|
+
from .domains import DOMAIN_NAMES, domain_scenarios # local import (domains.py imports SCENARIOS)
|
|
106
|
+
|
|
107
|
+
if domain == "all":
|
|
108
|
+
return [n for d in DOMAIN_NAMES for n in scenario_names(d)]
|
|
109
|
+
if domain == "agriculture":
|
|
110
|
+
return list(SCENARIOS)
|
|
111
|
+
return [f"{domain}:{s}" for s in domain_scenarios(domain)]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def get_scenario(name: str, *, randomize: bool = False) -> EdgeEngineAwareConfig:
|
|
115
|
+
"""Return a fresh configuration for ``name``.
|
|
116
|
+
|
|
117
|
+
``name`` is an agricultural scenario (see :data:`SCENARIOS`) or a qualified
|
|
118
|
+
``domain:scenario`` of another domain, e.g. ``"industrial:degrading"``
|
|
119
|
+
(see ``domains.DOMAINS``).
|
|
120
|
+
"""
|
|
121
|
+
from .domains import DOMAIN_NAMES, domain_scenarios # local import (domains.py imports SCENARIOS)
|
|
122
|
+
|
|
123
|
+
domain, scen = split_scenario_name(name)
|
|
124
|
+
if domain not in DOMAIN_NAMES:
|
|
125
|
+
raise KeyError(f"unknown domain {domain!r} in scenario {name!r}; available: {DOMAIN_NAMES}")
|
|
126
|
+
table = domain_scenarios(domain)
|
|
127
|
+
if scen not in table:
|
|
128
|
+
raise KeyError(f"unknown scenario {name!r}; available in {domain}: {sorted(table)}")
|
|
129
|
+
cfg = table[scen]()
|
|
130
|
+
cfg.randomization.enabled = randomize
|
|
131
|
+
return cfg
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Simulated scalar sensor (soil moisture, CO2, temperature - any normalised quantity).
|
|
2
|
+
|
|
3
|
+
The sensor reads the *true* field moisture and returns a noisy measurement
|
|
4
|
+
whose noise level depends on the requested sensing level. A real driver would
|
|
5
|
+
implement the same ``Sensor`` protocol on top of an ADC / I2C transaction.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Callable
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
|
|
14
|
+
from .config import SensingConfig
|
|
15
|
+
from .interfaces import Measurement
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class SimulatedSoilMoistureSensor:
|
|
19
|
+
"""``Sensor`` implementation backed by a ground-truth callable."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, cfg: SensingConfig, true_value: Callable[[], float]):
|
|
22
|
+
self.cfg = cfg
|
|
23
|
+
self._true_value = true_value
|
|
24
|
+
self._rng = np.random.default_rng()
|
|
25
|
+
|
|
26
|
+
def reset(self, rng: np.random.Generator) -> None:
|
|
27
|
+
self._rng = rng
|
|
28
|
+
|
|
29
|
+
# -- Sensor protocol ----------------------------------------------------
|
|
30
|
+
def energy_cost_j(self, level: int) -> float:
|
|
31
|
+
return self.cfg.energy_j[level]
|
|
32
|
+
|
|
33
|
+
def noise_std(self, level: int) -> float:
|
|
34
|
+
return self.cfg.noise_std[level]
|
|
35
|
+
|
|
36
|
+
def read(self, level: int, now_s: float) -> Measurement:
|
|
37
|
+
if level < 1 or level >= self.cfg.n_levels:
|
|
38
|
+
raise ValueError(f"sensing level must be in [1, {self.cfg.n_levels}), got {level}")
|
|
39
|
+
truth = self._true_value()
|
|
40
|
+
noise = self._rng.normal(0.0, self.cfg.noise_std[level]) + self.cfg.bias[level]
|
|
41
|
+
value = float(np.clip(truth + noise, 0.0, 1.0))
|
|
42
|
+
return Measurement(value=value, timestamp_s=now_s, level=level, noise_std=self.cfg.noise_std[level])
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
SimulatedScalarSensor = SimulatedSoilMoistureSensor # domain-neutral name
|