spicefault 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. spicefault/__init__.py +36 -0
  2. spicefault/circuit.py +76 -0
  3. spicefault/conditions/__init__.py +5 -0
  4. spicefault/conditions/operating.py +42 -0
  5. spicefault/dataset/__init__.py +19 -0
  6. spicefault/dataset/dataset.py +289 -0
  7. spicefault/dataset/manifest.py +79 -0
  8. spicefault/dataset/store.py +53 -0
  9. spicefault/experiments/__init__.py +21 -0
  10. spicefault/experiments/campaign.py +293 -0
  11. spicefault/experiments/engine.py +213 -0
  12. spicefault/experiments/experiment.py +305 -0
  13. spicefault/experiments/seeding.py +44 -0
  14. spicefault/faults/__init__.py +35 -0
  15. spicefault/faults/base.py +166 -0
  16. spicefault/faults/faultset.py +177 -0
  17. spicefault/faults/severity.py +61 -0
  18. spicefault/faults/types.py +244 -0
  19. spicefault/faults/universe.py +162 -0
  20. spicefault/measurements/__init__.py +7 -0
  21. spicefault/measurements/acquisition.py +12 -0
  22. spicefault/measurements/measurement.py +294 -0
  23. spicefault/measurements/response.py +18 -0
  24. spicefault/netlist.py +263 -0
  25. spicefault/reliability/__init__.py +56 -0
  26. spicefault/reliability/analysis.py +522 -0
  27. spicefault/reliability/detection.py +42 -0
  28. spicefault/reliability/sensitivity.py +104 -0
  29. spicefault/reliability/statistics.py +88 -0
  30. spicefault/reliability/structure.py +90 -0
  31. spicefault/simulation/__init__.py +39 -0
  32. spicefault/simulation/backend.py +213 -0
  33. spicefault/simulation/ngspice.py +125 -0
  34. spicefault/variation/__init__.py +36 -0
  35. spicefault/variation/base.py +123 -0
  36. spicefault/variation/distributions.py +35 -0
  37. spicefault/variation/types.py +324 -0
  38. spicefault-0.1.0.dist-info/METADATA +217 -0
  39. spicefault-0.1.0.dist-info/RECORD +41 -0
  40. spicefault-0.1.0.dist-info/WHEEL +4 -0
  41. spicefault-0.1.0.dist-info/licenses/LICENSE +21 -0
spicefault/__init__.py ADDED
@@ -0,0 +1,36 @@
1
+ """Reproducible SPICE-based fault injection and reliability assessment of electronic circuits."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ from .circuit import Circuit, Component # noqa: E402
6
+ from .conditions import OperatingCondition # noqa: E402
7
+ from .dataset import Dataset # noqa: E402
8
+ from .experiments import Experiment, ExperimentResult, FaultCampaign # noqa: E402
9
+ from .faults import Fault # noqa: E402
10
+ from .measurements import Measurement, Waveform # noqa: E402
11
+ from .simulation import ( # noqa: E402
12
+ SimulationConfig,
13
+ SimulationResult,
14
+ SimulationStatus,
15
+ Simulator,
16
+ )
17
+ from .variation import VariationSet # noqa: E402
18
+
19
+ __all__ = [
20
+ "Circuit",
21
+ "Component",
22
+ "Dataset",
23
+ "Experiment",
24
+ "ExperimentResult",
25
+ "Fault",
26
+ "FaultCampaign",
27
+ "Measurement",
28
+ "OperatingCondition",
29
+ "SimulationConfig",
30
+ "SimulationResult",
31
+ "SimulationStatus",
32
+ "Simulator",
33
+ "VariationSet",
34
+ "Waveform",
35
+ "__version__",
36
+ ]
spicefault/circuit.py ADDED
@@ -0,0 +1,76 @@
1
+ """The circuit under investigation: a SPICE netlist seen as components, nodes and parameters."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+
9
+ from .netlist import Netlist
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Component:
14
+ name: str
15
+ kind: str # SPICE element letter: R, C, L, V, X, ...
16
+ nodes: tuple[str, ...] # empty if the terminals of this element type are not known
17
+ parameters: dict[str, float] # numeric ones only: `value`, `dc`, instance parameters
18
+
19
+
20
+ class Circuit:
21
+ """A nominal circuit. It is never modified: every simulation works on a copy.
22
+
23
+ Only the top level of the netlist is exposed; what is inside a subcircuit is
24
+ reached through the parameters of its instances.
25
+ """
26
+
27
+ def __init__(self, netlist: str, name: str = "circuit"):
28
+ self.name = name
29
+ self._text = netlist
30
+
31
+ @classmethod
32
+ def from_netlist(cls, path: str | Path, name: str | None = None) -> Circuit:
33
+ path = Path(path)
34
+ return cls(path.read_text(), name or path.stem)
35
+
36
+ def to_netlist(self) -> str:
37
+ return self._text
38
+
39
+ def netlist(self) -> Netlist:
40
+ """A fresh, editable copy of the netlist."""
41
+ return Netlist(self._text)
42
+
43
+ def components(self) -> list[Component]:
44
+ net = self.netlist()
45
+ found = []
46
+ for name in net.components():
47
+ try:
48
+ nodes = tuple(net.nodes(name))
49
+ except NotImplementedError:
50
+ nodes = ()
51
+ found.append(Component(name, name[0].upper(), nodes, net.parameters(name)))
52
+ return found
53
+
54
+ def component(self, name: str) -> Component:
55
+ for component in self.components():
56
+ if component.name.lower() == name.lower():
57
+ return component
58
+ raise KeyError(f"no component named {name!r}")
59
+
60
+ def nodes(self) -> list[str]:
61
+ return sorted({node for c in self.components() for node in c.nodes})
62
+
63
+ def parameters(self) -> dict[tuple[str, str], float]:
64
+ """(component, parameter) -> nominal value, for every numeric parameter."""
65
+ return {
66
+ (c.name, parameter): value
67
+ for c in self.components()
68
+ for parameter, value in c.parameters.items()
69
+ }
70
+
71
+ def metadata(self) -> dict:
72
+ digest = hashlib.sha256(self._text.encode()).hexdigest()
73
+ return {"circuit": self.name, "netlist_sha256": digest}
74
+
75
+ def __repr__(self) -> str:
76
+ return f"Circuit({self.name!r}, {len(self.components())} components)"
@@ -0,0 +1,5 @@
1
+ """Operating conditions, kept apart from component uncertainty."""
2
+
3
+ from .operating import OperatingCondition
4
+
5
+ __all__ = ["OperatingCondition"]
@@ -0,0 +1,42 @@
1
+ """Conditions under which a circuit operates or is tested."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+
7
+ from ..netlist import Netlist
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class OperatingCondition:
12
+ """A named set of deterministic settings, applied after the fault.
13
+
14
+ `settings` maps (component, parameter) to an absolute value, for example
15
+ `{("Vcc", "dc"): 3.0}` for a supply or `{("Rload", "value"): 1e3}` for a load.
16
+ `temperature` is the simulation temperature in degrees Celsius; it only has an
17
+ effect on devices whose model depends on temperature.
18
+ """
19
+
20
+ name: str = "nominal"
21
+ temperature: float | None = None
22
+ settings: dict[tuple[str, str], float] = field(default_factory=dict)
23
+
24
+ def apply(self, netlist: Netlist) -> None:
25
+ for (component, parameter), value in self.settings.items():
26
+ netlist.set_parameter(component, parameter, "absolute", value)
27
+ if self.temperature is not None:
28
+ netlist.add_directive(f".options temp={float(self.temperature)}")
29
+
30
+ def metadata(self) -> dict:
31
+ return {
32
+ "name": self.name,
33
+ "temperature": self.temperature,
34
+ "settings": [
35
+ {"component": c, "parameter": p, "value": v} for (c, p), v in self.settings.items()
36
+ ],
37
+ }
38
+
39
+ @classmethod
40
+ def from_metadata(cls, record: dict) -> OperatingCondition:
41
+ settings = {(s["component"], s["parameter"]): s["value"] for s in record["settings"]}
42
+ return cls(record["name"], record["temperature"], settings)
@@ -0,0 +1,19 @@
1
+ """Dataset on disk: one of the representations of the results of a campaign."""
2
+
3
+ from .dataset import Dataset, Provenance
4
+ from .manifest import MANIFEST, Manifest, file_record
5
+ from .store import METADATA, SAMPLES, WAVEFORMS, assemble, load_dataset, load_metadata
6
+
7
+ __all__ = [
8
+ "MANIFEST",
9
+ "METADATA",
10
+ "SAMPLES",
11
+ "WAVEFORMS",
12
+ "Dataset",
13
+ "Manifest",
14
+ "Provenance",
15
+ "assemble",
16
+ "file_record",
17
+ "load_dataset",
18
+ "load_metadata",
19
+ ]
@@ -0,0 +1,289 @@
1
+ """The dataset of a campaign as an object: traceable, verifiable and reproducible."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ from collections.abc import Callable, Sequence
7
+ from dataclasses import asdict, dataclass
8
+ from pathlib import Path
9
+
10
+ import numpy as np
11
+ import pandas as pd
12
+
13
+ from ..circuit import Circuit
14
+ from ..conditions import OperatingCondition
15
+ from ..faults import FaultSet
16
+ from ..variation import VariationSet
17
+ from .manifest import Manifest
18
+ from .store import SAMPLES, WAVEFORMS, load_metadata
19
+
20
+ CIRCUIT_FILE = "circuit.cir"
21
+ HEALTHY_ID = "healthy"
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class Provenance:
26
+ """Everything that identifies one sample and what produced it."""
27
+
28
+ sample_id: int
29
+ seed: int
30
+ seeding: str
31
+ seed_key: str
32
+ circuit: str
33
+ netlist_sha256: str
34
+ fault_id: str
35
+ fault_type: str
36
+ fault_location: str
37
+ fault_magnitude: float | None
38
+ fault_severity: float | None
39
+ condition: str
40
+ status: str
41
+ message: str
42
+ simulator: str
43
+ simulator_version: str
44
+ spicefault_version: str
45
+ parameters: dict[str, float]
46
+
47
+ def to_dict(self) -> dict:
48
+ return asdict(self)
49
+
50
+
51
+ class Dataset:
52
+ """A dataset folder written by `FaultCampaign`.
53
+
54
+ `samples` has one row per simulation, failed ones included; `waveforms`, if any,
55
+ is aligned with it row by row. `metadata` is the definition of the experiment and
56
+ `manifest` the record of the run.
57
+ """
58
+
59
+ def __init__(self, path: str | Path):
60
+ self.path = Path(path)
61
+ self.manifest = Manifest.read(self.path)
62
+ self.metadata = load_metadata(self.path)
63
+ self.samples = pd.read_parquet(self.path / SAMPLES)
64
+ file = self.path / WAVEFORMS
65
+ self.waveforms = np.load(file, mmap_mode="r") if file.exists() else None
66
+ columns = self.manifest.summary.get("columns", {})
67
+ self.features: list[str] = list(columns.get("measurements", []))
68
+ self.labels: list[str] = list(columns.get("labels", []))
69
+ self.parameters: dict[str, tuple[str, str]] = {
70
+ column: tuple(target) for column, target in columns.get("parameters", {}).items()
71
+ }
72
+
73
+ def __len__(self) -> int:
74
+ return len(self.samples)
75
+
76
+ def __repr__(self) -> str:
77
+ return (
78
+ f"Dataset({str(self.path)!r}: {len(self)} samples, "
79
+ f"{len(self.metadata['faults'])} faults, {len(self.features)} measurements)"
80
+ )
81
+
82
+ @property
83
+ def ok(self) -> np.ndarray:
84
+ """True for the samples whose simulation gave a usable result."""
85
+ return self.samples["sim_ok"].to_numpy(dtype=bool)
86
+
87
+ @property
88
+ def circuit(self) -> Circuit:
89
+ return Circuit((self.path / CIRCUIT_FILE).read_text(), self.metadata["circuit"])
90
+
91
+ @property
92
+ def faults(self) -> FaultSet:
93
+ return FaultSet.from_metadata(self.metadata["faults"])
94
+
95
+ # --- integrity ----------------------------------------------------------------------
96
+
97
+ def verify(self) -> list[str]:
98
+ """Problems with the files or their consistency; an empty list if there are none.
99
+
100
+ The files must match the fingerprints of the manifest, the table must have one
101
+ row per planned sample, and failed samples must carry no result.
102
+ """
103
+ problems = self.manifest.verify(self.path)
104
+ df = self.samples
105
+ if len(df) != self.manifest.n_samples:
106
+ problems.append(f"{len(df)} rows, the manifest says {self.manifest.n_samples}")
107
+ if list(df["sample_id"]) != list(range(len(df))):
108
+ problems.append("sample_id is not the row number")
109
+ netlist = (self.path / CIRCUIT_FILE).read_bytes()
110
+ if hashlib.sha256(netlist).hexdigest() != self.metadata["netlist_sha256"]:
111
+ problems.append(f"{CIRCUIT_FILE} is not the netlist of the experiment")
112
+ defined = {HEALTHY_ID, *(f["fault_id"] for f in self.metadata["faults"])}
113
+ unknown = set(df["fault_id"]) - defined
114
+ if unknown:
115
+ problems.append(f"faults not defined in the metadata: {sorted(unknown)}")
116
+ ok = self.ok
117
+ if self.features and not np.isfinite(df.loc[ok, self.features].to_numpy(float)).all():
118
+ problems.append("a successful sample has a measurement that is not finite")
119
+ if self.features and df.loc[~ok, self.features].notna().any().any():
120
+ problems.append("a failed sample has a measurement")
121
+ if self.waveforms is not None:
122
+ if len(self.waveforms) != len(df):
123
+ problems.append("waveforms and samples have different lengths")
124
+ elif np.isnan(self.waveforms[ok]).any() or not np.isnan(self.waveforms[~ok]).all():
125
+ problems.append("waveforms do not match the status of the samples")
126
+ return problems
127
+
128
+ # --- traceability -------------------------------------------------------------------
129
+
130
+ def _row(self, sample_id: int) -> pd.Series:
131
+ row = self.samples.iloc[sample_id]
132
+ if row["sample_id"] != sample_id:
133
+ raise ValueError("sample_id is not the row number: the table was altered")
134
+ return row
135
+
136
+ def provenance(self, sample_id: int) -> Provenance:
137
+ row, meta = self._row(sample_id), self.metadata
138
+
139
+ def number(value) -> float | None:
140
+ return None if pd.isna(value) else float(value)
141
+
142
+ return Provenance(
143
+ sample_id=int(sample_id),
144
+ seed=meta["seed"],
145
+ seeding=meta["seeding"],
146
+ seed_key=row["seed_key"],
147
+ circuit=meta["circuit"],
148
+ netlist_sha256=meta["netlist_sha256"],
149
+ fault_id=row["fault_id"],
150
+ fault_type=row["fault_type"],
151
+ fault_location=row["fault_location"],
152
+ fault_magnitude=number(row["fault_magnitude"]),
153
+ fault_severity=number(row["fault_severity"]),
154
+ condition=row["condition"],
155
+ status=row["status"],
156
+ message=row["message"],
157
+ simulator=meta["simulator"],
158
+ simulator_version=meta["simulator_version"],
159
+ spicefault_version=meta["spicefault_version"],
160
+ parameters={c: float(row[c]) for c in self.parameters if not pd.isna(row[c])},
161
+ )
162
+
163
+ def netlist(self, sample_id: int) -> str:
164
+ """The netlist that was simulated for a sample, rebuilt from what is stored.
165
+
166
+ The realised values are read from the table, not drawn again, so this works
167
+ whatever variations the experiment used.
168
+ """
169
+ row = self._row(sample_id)
170
+ values = {target: row[column] for column, target in self.parameters.items()}
171
+ if any(pd.isna(v) for v in values.values()):
172
+ raise ValueError(f"sample {sample_id} has no realised values: {row['message']}")
173
+ netlist = self.circuit.netlist()
174
+ VariationSet.apply(netlist, {target: float(v) for target, v in values.items()})
175
+ if row["fault_id"] != HEALTHY_ID:
176
+ self.faults[row["fault_id"]].apply(netlist)
177
+ conditions = {c["name"]: c for c in self.metadata["conditions"]}
178
+ OperatingCondition.from_metadata(conditions[row["condition"]]).apply(netlist)
179
+ return str(netlist)
180
+
181
+ # --- reproducibility ----------------------------------------------------------------
182
+
183
+ def experiment(self, **overrides):
184
+ """The experiment that produced the dataset, rebuilt from its metadata.
185
+
186
+ Custom or joint variations, custom measurements and custom backends cannot be
187
+ stored and must be passed again (`variations=`, `measurements=`, `simulator=`).
188
+ """
189
+ from ..experiments import Experiment
190
+
191
+ return Experiment.from_metadata(
192
+ self.metadata, (self.path / CIRCUIT_FILE).read_text(), **overrides
193
+ )
194
+
195
+ def reproduce(
196
+ self,
197
+ sample_ids: Sequence[int] | None = None,
198
+ n: int = 20,
199
+ experiment=None,
200
+ seed: int = 0,
201
+ ) -> pd.DataFrame:
202
+ """Simulate samples again and compare them with what is stored.
203
+
204
+ By default `n` samples chosen at random. One row per sample:
205
+ - `definition`: same fault, condition and seed key;
206
+ - `parameters`: the redrawn values are exactly the stored ones;
207
+ - `status`: same simulation status;
208
+ - `max_abs_diff`, `max_rel_diff`: largest difference over the measurements;
209
+ - `waveform_abs_diff`: largest difference over the waveform, if stored.
210
+ On the platform and simulator version that wrote the dataset the differences
211
+ are expected to be zero; elsewhere they measure the dependence on both.
212
+ """
213
+ from ..experiments import simulate_sample
214
+
215
+ experiment = experiment or self.experiment()
216
+ plan = experiment.plan()
217
+ if len(plan) != len(self):
218
+ raise ValueError("the experiment does not have the samples of this dataset")
219
+ if sample_ids is None:
220
+ rng = np.random.default_rng(seed)
221
+ sample_ids = np.sort(rng.choice(len(self), min(n, len(self)), replace=False))
222
+ rows = []
223
+ for sample_id in map(int, sample_ids):
224
+ stored = self._row(sample_id)
225
+ new, waveform = simulate_sample(plan[sample_id], experiment)
226
+ ours = np.array([new[f] for f in self.features], dtype=float)
227
+ theirs = stored[self.features].to_numpy(dtype=float)
228
+ diff = np.abs(ours - theirs)
229
+ both_missing = np.isnan(ours) & np.isnan(theirs)
230
+ diff = np.where(both_missing, 0.0, diff)
231
+ waveform_diff = np.nan
232
+ if self.waveforms is not None and waveform is not None:
233
+ waveform_diff = float(np.abs(waveform - self.waveforms[sample_id]).max())
234
+ rows.append(
235
+ {
236
+ "sample_id": sample_id,
237
+ "definition": all(
238
+ new[c] == stored[c] for c in ("fault_id", "condition", "seed_key")
239
+ ),
240
+ "parameters": all(
241
+ new.get(c) == stored[c] or (pd.isna(new.get(c)) and pd.isna(stored[c]))
242
+ for c in self.parameters
243
+ ),
244
+ "status": new["status"] == stored["status"],
245
+ "max_abs_diff": float(diff.max()) if len(diff) else 0.0,
246
+ "max_rel_diff": float(
247
+ (diff / np.maximum(np.abs(theirs), 1e-300))[~both_missing].max()
248
+ )
249
+ if (~both_missing).any()
250
+ else 0.0,
251
+ "waveform_abs_diff": waveform_diff,
252
+ }
253
+ )
254
+ return pd.DataFrame(rows)
255
+
256
+ # --- uses ---------------------------------------------------------------------------
257
+
258
+ def to_ml(
259
+ self,
260
+ target: str | Callable[[pd.DataFrame], np.ndarray] = "fault_id",
261
+ features: Sequence[str] | None = None,
262
+ waveforms: bool = False,
263
+ drop_failed: bool = True,
264
+ ) -> tuple[np.ndarray, np.ndarray]:
265
+ """(X, y) for a statistical or machine-learning model.
266
+
267
+ X is the table of measurements (or of `features`), or the waveforms with
268
+ `waveforms`. y is a column of the samples, by default the fault identifier
269
+ (`fault_type` and `fault_location` are the usual alternatives), or the result
270
+ of a function of the samples. The library does not depend on any ML framework.
271
+ """
272
+ keep = self.ok if drop_failed else np.ones(len(self), dtype=bool)
273
+ samples = self.samples[keep]
274
+ if waveforms:
275
+ if self.waveforms is None:
276
+ raise ValueError("this dataset has no waveforms")
277
+ x = np.asarray(self.waveforms[keep])
278
+ else:
279
+ x = samples[list(features or self.features)].to_numpy(dtype=float)
280
+ y = target(samples) if callable(target) else samples[target].to_numpy()
281
+ return x, np.asarray(y)
282
+
283
+ def analysis(self, features: Sequence[str] | None = None, **kwargs):
284
+ """A `ReliabilityAnalysis` of the dataset."""
285
+ from ..reliability import ReliabilityAnalysis
286
+
287
+ return ReliabilityAnalysis(
288
+ self.samples, features or self.features, faults=self.metadata["faults"], **kwargs
289
+ )
@@ -0,0 +1,79 @@
1
+ """The manifest of a dataset: the record of the run that produced it."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from dataclasses import asdict, dataclass, field, fields
8
+ from pathlib import Path
9
+
10
+ SCHEMA_VERSION = 1
11
+ MANIFEST = "manifest.json"
12
+
13
+
14
+ def file_record(path: Path) -> dict:
15
+ """Size and SHA-256 of a file, read in blocks."""
16
+ digest = hashlib.sha256()
17
+ with path.open("rb") as stream:
18
+ for block in iter(lambda: stream.read(1 << 20), b""):
19
+ digest.update(block)
20
+ return {"bytes": path.stat().st_size, "sha256": digest.hexdigest()}
21
+
22
+
23
+ @dataclass
24
+ class Manifest:
25
+ """How and with what a dataset was produced, and the fingerprint of its files.
26
+
27
+ `summary` holds the counts the application adds (completed, failed, status
28
+ counts); in the file they are written next to the other fields. The manifest is
29
+ written last: its presence marks a complete dataset.
30
+ """
31
+
32
+ created: str
33
+ spicefault_version: str
34
+ simulator: str
35
+ python: str
36
+ numpy: str
37
+ pandas: str
38
+ platform: str
39
+ n_samples: int
40
+ workers: int
41
+ chunk: int
42
+ resumed: bool
43
+ elapsed_last_run_s: float
44
+ elapsed_total_s: float
45
+ files: dict[str, dict] = field(default_factory=dict)
46
+ summary: dict = field(default_factory=dict)
47
+ schema_version: int = SCHEMA_VERSION
48
+
49
+ def to_dict(self) -> dict:
50
+ record = asdict(self)
51
+ return {**{k: v for k, v in record.items() if k != "summary"}, **record["summary"]}
52
+
53
+ @classmethod
54
+ def from_dict(cls, record: dict) -> Manifest:
55
+ known = {f.name for f in fields(cls)} - {"summary"}
56
+ if record.get("schema_version") != SCHEMA_VERSION:
57
+ raise ValueError(f"unsupported manifest schema version {record.get('schema_version')}")
58
+ return cls(
59
+ **{k: v for k, v in record.items() if k in known},
60
+ summary={k: v for k, v in record.items() if k not in known},
61
+ )
62
+
63
+ def write(self, folder: str | Path) -> None:
64
+ (Path(folder) / MANIFEST).write_text(json.dumps(self.to_dict(), indent=2))
65
+
66
+ @classmethod
67
+ def read(cls, folder: str | Path) -> Manifest:
68
+ return cls.from_dict(json.loads((Path(folder) / MANIFEST).read_text()))
69
+
70
+ def verify(self, folder: str | Path) -> list[str]:
71
+ """Files that are missing or no longer match their recorded fingerprint."""
72
+ problems = []
73
+ for name, expected in self.files.items():
74
+ path = Path(folder) / name
75
+ if not path.exists():
76
+ problems.append(f"{name}: missing")
77
+ elif file_record(path) != expected:
78
+ problems.append(f"{name}: content differs from the manifest")
79
+ return problems
@@ -0,0 +1,53 @@
1
+ """Reading and writing of a dataset directory.
2
+
3
+ A dataset is a directory with:
4
+ - samples.parquet one row per simulation;
5
+ - waveforms.npy float32 [n_samples, n_points], row-aligned with the table (optional);
6
+ - metadata.json the definition of what was simulated: enough to regenerate it;
7
+ - manifest.json the record of the run: versions, counts, fingerprints of the files;
8
+ - any other file the application adds, such as the source netlist.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ from pathlib import Path
15
+
16
+ import numpy as np
17
+ import pandas as pd
18
+
19
+ from .manifest import MANIFEST
20
+
21
+ SAMPLES, WAVEFORMS, METADATA = "samples.parquet", "waveforms.npy", "metadata.json"
22
+
23
+
24
+ def assemble(stems: list[Path], out_dir: str | Path) -> tuple[pd.DataFrame, np.ndarray | None]:
25
+ """Join the chunks written by a campaign into `out_dir`; returns (samples, waveforms)."""
26
+ out_dir = Path(out_dir)
27
+ df = pd.concat([pd.read_parquet(s.with_suffix(".parquet")) for s in stems], ignore_index=True)
28
+ df.to_parquet(out_dir / SAMPLES, index=False)
29
+ if not all(s.with_suffix(".npy").exists() for s in stems):
30
+ return df, None
31
+ waveforms = np.concatenate([np.load(s.with_suffix(".npy")) for s in stems])
32
+ np.save(out_dir / WAVEFORMS, waveforms)
33
+ return df, waveforms
34
+
35
+
36
+ def load_metadata(path: str | Path) -> dict:
37
+ """The definition of the campaign that wrote the dataset."""
38
+ return json.loads((Path(path) / METADATA).read_text())
39
+
40
+
41
+ def load_dataset(
42
+ path: str | Path, drop_failed: bool = True, ok_column: str = "sim_ok"
43
+ ) -> tuple[pd.DataFrame, np.ndarray | None, dict]:
44
+ """Return (samples, waveforms, manifest). Failed simulations are dropped on request."""
45
+ path = Path(path)
46
+ df = pd.read_parquet(path / SAMPLES)
47
+ waveforms = np.load(path / WAVEFORMS) if (path / WAVEFORMS).exists() else None
48
+ manifest = json.loads((path / MANIFEST).read_text())
49
+ if drop_failed and ok_column in df:
50
+ keep = df[ok_column].to_numpy()
51
+ df = df[keep].reset_index(drop=True)
52
+ waveforms = waveforms[keep] if waveforms is not None else None
53
+ return df, waveforms, manifest
@@ -0,0 +1,21 @@
1
+ """Experiments: deterministic sampling and parallel, resumable execution."""
2
+
3
+ from .campaign import FaultCampaign, ValidationReport, simulate_sample
4
+ from .engine import config_key, run_campaign, run_chunks
5
+ from .experiment import Experiment, ExperimentResult, Realisation, Sample, SampleResult
6
+ from .seeding import sample_stream
7
+
8
+ __all__ = [
9
+ "Experiment",
10
+ "ExperimentResult",
11
+ "FaultCampaign",
12
+ "Realisation",
13
+ "Sample",
14
+ "SampleResult",
15
+ "ValidationReport",
16
+ "config_key",
17
+ "run_campaign",
18
+ "run_chunks",
19
+ "sample_stream",
20
+ "simulate_sample",
21
+ ]