spicefault 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spicefault/__init__.py +36 -0
- spicefault/circuit.py +76 -0
- spicefault/conditions/__init__.py +5 -0
- spicefault/conditions/operating.py +42 -0
- spicefault/dataset/__init__.py +19 -0
- spicefault/dataset/dataset.py +289 -0
- spicefault/dataset/manifest.py +79 -0
- spicefault/dataset/store.py +53 -0
- spicefault/experiments/__init__.py +21 -0
- spicefault/experiments/campaign.py +293 -0
- spicefault/experiments/engine.py +213 -0
- spicefault/experiments/experiment.py +305 -0
- spicefault/experiments/seeding.py +44 -0
- spicefault/faults/__init__.py +35 -0
- spicefault/faults/base.py +166 -0
- spicefault/faults/faultset.py +177 -0
- spicefault/faults/severity.py +61 -0
- spicefault/faults/types.py +244 -0
- spicefault/faults/universe.py +162 -0
- spicefault/measurements/__init__.py +7 -0
- spicefault/measurements/acquisition.py +12 -0
- spicefault/measurements/measurement.py +294 -0
- spicefault/measurements/response.py +18 -0
- spicefault/netlist.py +263 -0
- spicefault/reliability/__init__.py +56 -0
- spicefault/reliability/analysis.py +522 -0
- spicefault/reliability/detection.py +42 -0
- spicefault/reliability/sensitivity.py +104 -0
- spicefault/reliability/statistics.py +88 -0
- spicefault/reliability/structure.py +90 -0
- spicefault/simulation/__init__.py +39 -0
- spicefault/simulation/backend.py +213 -0
- spicefault/simulation/ngspice.py +125 -0
- spicefault/variation/__init__.py +36 -0
- spicefault/variation/base.py +123 -0
- spicefault/variation/distributions.py +35 -0
- spicefault/variation/types.py +324 -0
- spicefault-0.1.0.dist-info/METADATA +217 -0
- spicefault-0.1.0.dist-info/RECORD +41 -0
- spicefault-0.1.0.dist-info/WHEEL +4 -0
- spicefault-0.1.0.dist-info/licenses/LICENSE +21 -0
spicefault/__init__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Reproducible SPICE-based fault injection and reliability assessment of electronic circuits."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
from .circuit import Circuit, Component # noqa: E402
|
|
6
|
+
from .conditions import OperatingCondition # noqa: E402
|
|
7
|
+
from .dataset import Dataset # noqa: E402
|
|
8
|
+
from .experiments import Experiment, ExperimentResult, FaultCampaign # noqa: E402
|
|
9
|
+
from .faults import Fault # noqa: E402
|
|
10
|
+
from .measurements import Measurement, Waveform # noqa: E402
|
|
11
|
+
from .simulation import ( # noqa: E402
|
|
12
|
+
SimulationConfig,
|
|
13
|
+
SimulationResult,
|
|
14
|
+
SimulationStatus,
|
|
15
|
+
Simulator,
|
|
16
|
+
)
|
|
17
|
+
from .variation import VariationSet # noqa: E402
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Circuit",
|
|
21
|
+
"Component",
|
|
22
|
+
"Dataset",
|
|
23
|
+
"Experiment",
|
|
24
|
+
"ExperimentResult",
|
|
25
|
+
"Fault",
|
|
26
|
+
"FaultCampaign",
|
|
27
|
+
"Measurement",
|
|
28
|
+
"OperatingCondition",
|
|
29
|
+
"SimulationConfig",
|
|
30
|
+
"SimulationResult",
|
|
31
|
+
"SimulationStatus",
|
|
32
|
+
"Simulator",
|
|
33
|
+
"VariationSet",
|
|
34
|
+
"Waveform",
|
|
35
|
+
"__version__",
|
|
36
|
+
]
|
spicefault/circuit.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""The circuit under investigation: a SPICE netlist seen as components, nodes and parameters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from .netlist import Netlist
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Component:
|
|
14
|
+
name: str
|
|
15
|
+
kind: str # SPICE element letter: R, C, L, V, X, ...
|
|
16
|
+
nodes: tuple[str, ...] # empty if the terminals of this element type are not known
|
|
17
|
+
parameters: dict[str, float] # numeric ones only: `value`, `dc`, instance parameters
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Circuit:
|
|
21
|
+
"""A nominal circuit. It is never modified: every simulation works on a copy.
|
|
22
|
+
|
|
23
|
+
Only the top level of the netlist is exposed; what is inside a subcircuit is
|
|
24
|
+
reached through the parameters of its instances.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(self, netlist: str, name: str = "circuit"):
|
|
28
|
+
self.name = name
|
|
29
|
+
self._text = netlist
|
|
30
|
+
|
|
31
|
+
@classmethod
|
|
32
|
+
def from_netlist(cls, path: str | Path, name: str | None = None) -> Circuit:
|
|
33
|
+
path = Path(path)
|
|
34
|
+
return cls(path.read_text(), name or path.stem)
|
|
35
|
+
|
|
36
|
+
def to_netlist(self) -> str:
|
|
37
|
+
return self._text
|
|
38
|
+
|
|
39
|
+
def netlist(self) -> Netlist:
|
|
40
|
+
"""A fresh, editable copy of the netlist."""
|
|
41
|
+
return Netlist(self._text)
|
|
42
|
+
|
|
43
|
+
def components(self) -> list[Component]:
|
|
44
|
+
net = self.netlist()
|
|
45
|
+
found = []
|
|
46
|
+
for name in net.components():
|
|
47
|
+
try:
|
|
48
|
+
nodes = tuple(net.nodes(name))
|
|
49
|
+
except NotImplementedError:
|
|
50
|
+
nodes = ()
|
|
51
|
+
found.append(Component(name, name[0].upper(), nodes, net.parameters(name)))
|
|
52
|
+
return found
|
|
53
|
+
|
|
54
|
+
def component(self, name: str) -> Component:
|
|
55
|
+
for component in self.components():
|
|
56
|
+
if component.name.lower() == name.lower():
|
|
57
|
+
return component
|
|
58
|
+
raise KeyError(f"no component named {name!r}")
|
|
59
|
+
|
|
60
|
+
def nodes(self) -> list[str]:
|
|
61
|
+
return sorted({node for c in self.components() for node in c.nodes})
|
|
62
|
+
|
|
63
|
+
def parameters(self) -> dict[tuple[str, str], float]:
|
|
64
|
+
"""(component, parameter) -> nominal value, for every numeric parameter."""
|
|
65
|
+
return {
|
|
66
|
+
(c.name, parameter): value
|
|
67
|
+
for c in self.components()
|
|
68
|
+
for parameter, value in c.parameters.items()
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
def metadata(self) -> dict:
|
|
72
|
+
digest = hashlib.sha256(self._text.encode()).hexdigest()
|
|
73
|
+
return {"circuit": self.name, "netlist_sha256": digest}
|
|
74
|
+
|
|
75
|
+
def __repr__(self) -> str:
|
|
76
|
+
return f"Circuit({self.name!r}, {len(self.components())} components)"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Conditions under which a circuit operates or is tested."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from ..netlist import Netlist
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class OperatingCondition:
|
|
12
|
+
"""A named set of deterministic settings, applied after the fault.
|
|
13
|
+
|
|
14
|
+
`settings` maps (component, parameter) to an absolute value, for example
|
|
15
|
+
`{("Vcc", "dc"): 3.0}` for a supply or `{("Rload", "value"): 1e3}` for a load.
|
|
16
|
+
`temperature` is the simulation temperature in degrees Celsius; it only has an
|
|
17
|
+
effect on devices whose model depends on temperature.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
name: str = "nominal"
|
|
21
|
+
temperature: float | None = None
|
|
22
|
+
settings: dict[tuple[str, str], float] = field(default_factory=dict)
|
|
23
|
+
|
|
24
|
+
def apply(self, netlist: Netlist) -> None:
|
|
25
|
+
for (component, parameter), value in self.settings.items():
|
|
26
|
+
netlist.set_parameter(component, parameter, "absolute", value)
|
|
27
|
+
if self.temperature is not None:
|
|
28
|
+
netlist.add_directive(f".options temp={float(self.temperature)}")
|
|
29
|
+
|
|
30
|
+
def metadata(self) -> dict:
|
|
31
|
+
return {
|
|
32
|
+
"name": self.name,
|
|
33
|
+
"temperature": self.temperature,
|
|
34
|
+
"settings": [
|
|
35
|
+
{"component": c, "parameter": p, "value": v} for (c, p), v in self.settings.items()
|
|
36
|
+
],
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def from_metadata(cls, record: dict) -> OperatingCondition:
|
|
41
|
+
settings = {(s["component"], s["parameter"]): s["value"] for s in record["settings"]}
|
|
42
|
+
return cls(record["name"], record["temperature"], settings)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Dataset on disk: one of the representations of the results of a campaign."""
|
|
2
|
+
|
|
3
|
+
from .dataset import Dataset, Provenance
|
|
4
|
+
from .manifest import MANIFEST, Manifest, file_record
|
|
5
|
+
from .store import METADATA, SAMPLES, WAVEFORMS, assemble, load_dataset, load_metadata
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"MANIFEST",
|
|
9
|
+
"METADATA",
|
|
10
|
+
"SAMPLES",
|
|
11
|
+
"WAVEFORMS",
|
|
12
|
+
"Dataset",
|
|
13
|
+
"Manifest",
|
|
14
|
+
"Provenance",
|
|
15
|
+
"assemble",
|
|
16
|
+
"file_record",
|
|
17
|
+
"load_dataset",
|
|
18
|
+
"load_metadata",
|
|
19
|
+
]
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""The dataset of a campaign as an object: traceable, verifiable and reproducible."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
from collections.abc import Callable, Sequence
|
|
7
|
+
from dataclasses import asdict, dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
from ..circuit import Circuit
|
|
14
|
+
from ..conditions import OperatingCondition
|
|
15
|
+
from ..faults import FaultSet
|
|
16
|
+
from ..variation import VariationSet
|
|
17
|
+
from .manifest import Manifest
|
|
18
|
+
from .store import SAMPLES, WAVEFORMS, load_metadata
|
|
19
|
+
|
|
20
|
+
CIRCUIT_FILE = "circuit.cir"
|
|
21
|
+
HEALTHY_ID = "healthy"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class Provenance:
|
|
26
|
+
"""Everything that identifies one sample and what produced it."""
|
|
27
|
+
|
|
28
|
+
sample_id: int
|
|
29
|
+
seed: int
|
|
30
|
+
seeding: str
|
|
31
|
+
seed_key: str
|
|
32
|
+
circuit: str
|
|
33
|
+
netlist_sha256: str
|
|
34
|
+
fault_id: str
|
|
35
|
+
fault_type: str
|
|
36
|
+
fault_location: str
|
|
37
|
+
fault_magnitude: float | None
|
|
38
|
+
fault_severity: float | None
|
|
39
|
+
condition: str
|
|
40
|
+
status: str
|
|
41
|
+
message: str
|
|
42
|
+
simulator: str
|
|
43
|
+
simulator_version: str
|
|
44
|
+
spicefault_version: str
|
|
45
|
+
parameters: dict[str, float]
|
|
46
|
+
|
|
47
|
+
def to_dict(self) -> dict:
|
|
48
|
+
return asdict(self)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class Dataset:
|
|
52
|
+
"""A dataset folder written by `FaultCampaign`.
|
|
53
|
+
|
|
54
|
+
`samples` has one row per simulation, failed ones included; `waveforms`, if any,
|
|
55
|
+
is aligned with it row by row. `metadata` is the definition of the experiment and
|
|
56
|
+
`manifest` the record of the run.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
def __init__(self, path: str | Path):
|
|
60
|
+
self.path = Path(path)
|
|
61
|
+
self.manifest = Manifest.read(self.path)
|
|
62
|
+
self.metadata = load_metadata(self.path)
|
|
63
|
+
self.samples = pd.read_parquet(self.path / SAMPLES)
|
|
64
|
+
file = self.path / WAVEFORMS
|
|
65
|
+
self.waveforms = np.load(file, mmap_mode="r") if file.exists() else None
|
|
66
|
+
columns = self.manifest.summary.get("columns", {})
|
|
67
|
+
self.features: list[str] = list(columns.get("measurements", []))
|
|
68
|
+
self.labels: list[str] = list(columns.get("labels", []))
|
|
69
|
+
self.parameters: dict[str, tuple[str, str]] = {
|
|
70
|
+
column: tuple(target) for column, target in columns.get("parameters", {}).items()
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
def __len__(self) -> int:
|
|
74
|
+
return len(self.samples)
|
|
75
|
+
|
|
76
|
+
def __repr__(self) -> str:
|
|
77
|
+
return (
|
|
78
|
+
f"Dataset({str(self.path)!r}: {len(self)} samples, "
|
|
79
|
+
f"{len(self.metadata['faults'])} faults, {len(self.features)} measurements)"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def ok(self) -> np.ndarray:
|
|
84
|
+
"""True for the samples whose simulation gave a usable result."""
|
|
85
|
+
return self.samples["sim_ok"].to_numpy(dtype=bool)
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def circuit(self) -> Circuit:
|
|
89
|
+
return Circuit((self.path / CIRCUIT_FILE).read_text(), self.metadata["circuit"])
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def faults(self) -> FaultSet:
|
|
93
|
+
return FaultSet.from_metadata(self.metadata["faults"])
|
|
94
|
+
|
|
95
|
+
# --- integrity ----------------------------------------------------------------------
|
|
96
|
+
|
|
97
|
+
def verify(self) -> list[str]:
|
|
98
|
+
"""Problems with the files or their consistency; an empty list if there are none.
|
|
99
|
+
|
|
100
|
+
The files must match the fingerprints of the manifest, the table must have one
|
|
101
|
+
row per planned sample, and failed samples must carry no result.
|
|
102
|
+
"""
|
|
103
|
+
problems = self.manifest.verify(self.path)
|
|
104
|
+
df = self.samples
|
|
105
|
+
if len(df) != self.manifest.n_samples:
|
|
106
|
+
problems.append(f"{len(df)} rows, the manifest says {self.manifest.n_samples}")
|
|
107
|
+
if list(df["sample_id"]) != list(range(len(df))):
|
|
108
|
+
problems.append("sample_id is not the row number")
|
|
109
|
+
netlist = (self.path / CIRCUIT_FILE).read_bytes()
|
|
110
|
+
if hashlib.sha256(netlist).hexdigest() != self.metadata["netlist_sha256"]:
|
|
111
|
+
problems.append(f"{CIRCUIT_FILE} is not the netlist of the experiment")
|
|
112
|
+
defined = {HEALTHY_ID, *(f["fault_id"] for f in self.metadata["faults"])}
|
|
113
|
+
unknown = set(df["fault_id"]) - defined
|
|
114
|
+
if unknown:
|
|
115
|
+
problems.append(f"faults not defined in the metadata: {sorted(unknown)}")
|
|
116
|
+
ok = self.ok
|
|
117
|
+
if self.features and not np.isfinite(df.loc[ok, self.features].to_numpy(float)).all():
|
|
118
|
+
problems.append("a successful sample has a measurement that is not finite")
|
|
119
|
+
if self.features and df.loc[~ok, self.features].notna().any().any():
|
|
120
|
+
problems.append("a failed sample has a measurement")
|
|
121
|
+
if self.waveforms is not None:
|
|
122
|
+
if len(self.waveforms) != len(df):
|
|
123
|
+
problems.append("waveforms and samples have different lengths")
|
|
124
|
+
elif np.isnan(self.waveforms[ok]).any() or not np.isnan(self.waveforms[~ok]).all():
|
|
125
|
+
problems.append("waveforms do not match the status of the samples")
|
|
126
|
+
return problems
|
|
127
|
+
|
|
128
|
+
# --- traceability -------------------------------------------------------------------
|
|
129
|
+
|
|
130
|
+
def _row(self, sample_id: int) -> pd.Series:
|
|
131
|
+
row = self.samples.iloc[sample_id]
|
|
132
|
+
if row["sample_id"] != sample_id:
|
|
133
|
+
raise ValueError("sample_id is not the row number: the table was altered")
|
|
134
|
+
return row
|
|
135
|
+
|
|
136
|
+
def provenance(self, sample_id: int) -> Provenance:
|
|
137
|
+
row, meta = self._row(sample_id), self.metadata
|
|
138
|
+
|
|
139
|
+
def number(value) -> float | None:
|
|
140
|
+
return None if pd.isna(value) else float(value)
|
|
141
|
+
|
|
142
|
+
return Provenance(
|
|
143
|
+
sample_id=int(sample_id),
|
|
144
|
+
seed=meta["seed"],
|
|
145
|
+
seeding=meta["seeding"],
|
|
146
|
+
seed_key=row["seed_key"],
|
|
147
|
+
circuit=meta["circuit"],
|
|
148
|
+
netlist_sha256=meta["netlist_sha256"],
|
|
149
|
+
fault_id=row["fault_id"],
|
|
150
|
+
fault_type=row["fault_type"],
|
|
151
|
+
fault_location=row["fault_location"],
|
|
152
|
+
fault_magnitude=number(row["fault_magnitude"]),
|
|
153
|
+
fault_severity=number(row["fault_severity"]),
|
|
154
|
+
condition=row["condition"],
|
|
155
|
+
status=row["status"],
|
|
156
|
+
message=row["message"],
|
|
157
|
+
simulator=meta["simulator"],
|
|
158
|
+
simulator_version=meta["simulator_version"],
|
|
159
|
+
spicefault_version=meta["spicefault_version"],
|
|
160
|
+
parameters={c: float(row[c]) for c in self.parameters if not pd.isna(row[c])},
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
def netlist(self, sample_id: int) -> str:
|
|
164
|
+
"""The netlist that was simulated for a sample, rebuilt from what is stored.
|
|
165
|
+
|
|
166
|
+
The realised values are read from the table, not drawn again, so this works
|
|
167
|
+
whatever variations the experiment used.
|
|
168
|
+
"""
|
|
169
|
+
row = self._row(sample_id)
|
|
170
|
+
values = {target: row[column] for column, target in self.parameters.items()}
|
|
171
|
+
if any(pd.isna(v) for v in values.values()):
|
|
172
|
+
raise ValueError(f"sample {sample_id} has no realised values: {row['message']}")
|
|
173
|
+
netlist = self.circuit.netlist()
|
|
174
|
+
VariationSet.apply(netlist, {target: float(v) for target, v in values.items()})
|
|
175
|
+
if row["fault_id"] != HEALTHY_ID:
|
|
176
|
+
self.faults[row["fault_id"]].apply(netlist)
|
|
177
|
+
conditions = {c["name"]: c for c in self.metadata["conditions"]}
|
|
178
|
+
OperatingCondition.from_metadata(conditions[row["condition"]]).apply(netlist)
|
|
179
|
+
return str(netlist)
|
|
180
|
+
|
|
181
|
+
# --- reproducibility ----------------------------------------------------------------
|
|
182
|
+
|
|
183
|
+
def experiment(self, **overrides):
|
|
184
|
+
"""The experiment that produced the dataset, rebuilt from its metadata.
|
|
185
|
+
|
|
186
|
+
Custom or joint variations, custom measurements and custom backends cannot be
|
|
187
|
+
stored and must be passed again (`variations=`, `measurements=`, `simulator=`).
|
|
188
|
+
"""
|
|
189
|
+
from ..experiments import Experiment
|
|
190
|
+
|
|
191
|
+
return Experiment.from_metadata(
|
|
192
|
+
self.metadata, (self.path / CIRCUIT_FILE).read_text(), **overrides
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
def reproduce(
|
|
196
|
+
self,
|
|
197
|
+
sample_ids: Sequence[int] | None = None,
|
|
198
|
+
n: int = 20,
|
|
199
|
+
experiment=None,
|
|
200
|
+
seed: int = 0,
|
|
201
|
+
) -> pd.DataFrame:
|
|
202
|
+
"""Simulate samples again and compare them with what is stored.
|
|
203
|
+
|
|
204
|
+
By default `n` samples chosen at random. One row per sample:
|
|
205
|
+
- `definition`: same fault, condition and seed key;
|
|
206
|
+
- `parameters`: the redrawn values are exactly the stored ones;
|
|
207
|
+
- `status`: same simulation status;
|
|
208
|
+
- `max_abs_diff`, `max_rel_diff`: largest difference over the measurements;
|
|
209
|
+
- `waveform_abs_diff`: largest difference over the waveform, if stored.
|
|
210
|
+
On the platform and simulator version that wrote the dataset the differences
|
|
211
|
+
are expected to be zero; elsewhere they measure the dependence on both.
|
|
212
|
+
"""
|
|
213
|
+
from ..experiments import simulate_sample
|
|
214
|
+
|
|
215
|
+
experiment = experiment or self.experiment()
|
|
216
|
+
plan = experiment.plan()
|
|
217
|
+
if len(plan) != len(self):
|
|
218
|
+
raise ValueError("the experiment does not have the samples of this dataset")
|
|
219
|
+
if sample_ids is None:
|
|
220
|
+
rng = np.random.default_rng(seed)
|
|
221
|
+
sample_ids = np.sort(rng.choice(len(self), min(n, len(self)), replace=False))
|
|
222
|
+
rows = []
|
|
223
|
+
for sample_id in map(int, sample_ids):
|
|
224
|
+
stored = self._row(sample_id)
|
|
225
|
+
new, waveform = simulate_sample(plan[sample_id], experiment)
|
|
226
|
+
ours = np.array([new[f] for f in self.features], dtype=float)
|
|
227
|
+
theirs = stored[self.features].to_numpy(dtype=float)
|
|
228
|
+
diff = np.abs(ours - theirs)
|
|
229
|
+
both_missing = np.isnan(ours) & np.isnan(theirs)
|
|
230
|
+
diff = np.where(both_missing, 0.0, diff)
|
|
231
|
+
waveform_diff = np.nan
|
|
232
|
+
if self.waveforms is not None and waveform is not None:
|
|
233
|
+
waveform_diff = float(np.abs(waveform - self.waveforms[sample_id]).max())
|
|
234
|
+
rows.append(
|
|
235
|
+
{
|
|
236
|
+
"sample_id": sample_id,
|
|
237
|
+
"definition": all(
|
|
238
|
+
new[c] == stored[c] for c in ("fault_id", "condition", "seed_key")
|
|
239
|
+
),
|
|
240
|
+
"parameters": all(
|
|
241
|
+
new.get(c) == stored[c] or (pd.isna(new.get(c)) and pd.isna(stored[c]))
|
|
242
|
+
for c in self.parameters
|
|
243
|
+
),
|
|
244
|
+
"status": new["status"] == stored["status"],
|
|
245
|
+
"max_abs_diff": float(diff.max()) if len(diff) else 0.0,
|
|
246
|
+
"max_rel_diff": float(
|
|
247
|
+
(diff / np.maximum(np.abs(theirs), 1e-300))[~both_missing].max()
|
|
248
|
+
)
|
|
249
|
+
if (~both_missing).any()
|
|
250
|
+
else 0.0,
|
|
251
|
+
"waveform_abs_diff": waveform_diff,
|
|
252
|
+
}
|
|
253
|
+
)
|
|
254
|
+
return pd.DataFrame(rows)
|
|
255
|
+
|
|
256
|
+
# --- uses ---------------------------------------------------------------------------
|
|
257
|
+
|
|
258
|
+
def to_ml(
|
|
259
|
+
self,
|
|
260
|
+
target: str | Callable[[pd.DataFrame], np.ndarray] = "fault_id",
|
|
261
|
+
features: Sequence[str] | None = None,
|
|
262
|
+
waveforms: bool = False,
|
|
263
|
+
drop_failed: bool = True,
|
|
264
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
265
|
+
"""(X, y) for a statistical or machine-learning model.
|
|
266
|
+
|
|
267
|
+
X is the table of measurements (or of `features`), or the waveforms with
|
|
268
|
+
`waveforms`. y is a column of the samples, by default the fault identifier
|
|
269
|
+
(`fault_type` and `fault_location` are the usual alternatives), or the result
|
|
270
|
+
of a function of the samples. The library does not depend on any ML framework.
|
|
271
|
+
"""
|
|
272
|
+
keep = self.ok if drop_failed else np.ones(len(self), dtype=bool)
|
|
273
|
+
samples = self.samples[keep]
|
|
274
|
+
if waveforms:
|
|
275
|
+
if self.waveforms is None:
|
|
276
|
+
raise ValueError("this dataset has no waveforms")
|
|
277
|
+
x = np.asarray(self.waveforms[keep])
|
|
278
|
+
else:
|
|
279
|
+
x = samples[list(features or self.features)].to_numpy(dtype=float)
|
|
280
|
+
y = target(samples) if callable(target) else samples[target].to_numpy()
|
|
281
|
+
return x, np.asarray(y)
|
|
282
|
+
|
|
283
|
+
def analysis(self, features: Sequence[str] | None = None, **kwargs):
|
|
284
|
+
"""A `ReliabilityAnalysis` of the dataset."""
|
|
285
|
+
from ..reliability import ReliabilityAnalysis
|
|
286
|
+
|
|
287
|
+
return ReliabilityAnalysis(
|
|
288
|
+
self.samples, features or self.features, faults=self.metadata["faults"], **kwargs
|
|
289
|
+
)
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""The manifest of a dataset: the record of the run that produced it."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from dataclasses import asdict, dataclass, field, fields
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
SCHEMA_VERSION = 1
|
|
11
|
+
MANIFEST = "manifest.json"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def file_record(path: Path) -> dict:
|
|
15
|
+
"""Size and SHA-256 of a file, read in blocks."""
|
|
16
|
+
digest = hashlib.sha256()
|
|
17
|
+
with path.open("rb") as stream:
|
|
18
|
+
for block in iter(lambda: stream.read(1 << 20), b""):
|
|
19
|
+
digest.update(block)
|
|
20
|
+
return {"bytes": path.stat().st_size, "sha256": digest.hexdigest()}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Manifest:
|
|
25
|
+
"""How and with what a dataset was produced, and the fingerprint of its files.
|
|
26
|
+
|
|
27
|
+
`summary` holds the counts the application adds (completed, failed, status
|
|
28
|
+
counts); in the file they are written next to the other fields. The manifest is
|
|
29
|
+
written last: its presence marks a complete dataset.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
created: str
|
|
33
|
+
spicefault_version: str
|
|
34
|
+
simulator: str
|
|
35
|
+
python: str
|
|
36
|
+
numpy: str
|
|
37
|
+
pandas: str
|
|
38
|
+
platform: str
|
|
39
|
+
n_samples: int
|
|
40
|
+
workers: int
|
|
41
|
+
chunk: int
|
|
42
|
+
resumed: bool
|
|
43
|
+
elapsed_last_run_s: float
|
|
44
|
+
elapsed_total_s: float
|
|
45
|
+
files: dict[str, dict] = field(default_factory=dict)
|
|
46
|
+
summary: dict = field(default_factory=dict)
|
|
47
|
+
schema_version: int = SCHEMA_VERSION
|
|
48
|
+
|
|
49
|
+
def to_dict(self) -> dict:
|
|
50
|
+
record = asdict(self)
|
|
51
|
+
return {**{k: v for k, v in record.items() if k != "summary"}, **record["summary"]}
|
|
52
|
+
|
|
53
|
+
@classmethod
|
|
54
|
+
def from_dict(cls, record: dict) -> Manifest:
|
|
55
|
+
known = {f.name for f in fields(cls)} - {"summary"}
|
|
56
|
+
if record.get("schema_version") != SCHEMA_VERSION:
|
|
57
|
+
raise ValueError(f"unsupported manifest schema version {record.get('schema_version')}")
|
|
58
|
+
return cls(
|
|
59
|
+
**{k: v for k, v in record.items() if k in known},
|
|
60
|
+
summary={k: v for k, v in record.items() if k not in known},
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
def write(self, folder: str | Path) -> None:
|
|
64
|
+
(Path(folder) / MANIFEST).write_text(json.dumps(self.to_dict(), indent=2))
|
|
65
|
+
|
|
66
|
+
@classmethod
|
|
67
|
+
def read(cls, folder: str | Path) -> Manifest:
|
|
68
|
+
return cls.from_dict(json.loads((Path(folder) / MANIFEST).read_text()))
|
|
69
|
+
|
|
70
|
+
def verify(self, folder: str | Path) -> list[str]:
|
|
71
|
+
"""Files that are missing or no longer match their recorded fingerprint."""
|
|
72
|
+
problems = []
|
|
73
|
+
for name, expected in self.files.items():
|
|
74
|
+
path = Path(folder) / name
|
|
75
|
+
if not path.exists():
|
|
76
|
+
problems.append(f"{name}: missing")
|
|
77
|
+
elif file_record(path) != expected:
|
|
78
|
+
problems.append(f"{name}: content differs from the manifest")
|
|
79
|
+
return problems
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Reading and writing of a dataset directory.
|
|
2
|
+
|
|
3
|
+
A dataset is a directory with:
|
|
4
|
+
- samples.parquet one row per simulation;
|
|
5
|
+
- waveforms.npy float32 [n_samples, n_points], row-aligned with the table (optional);
|
|
6
|
+
- metadata.json the definition of what was simulated: enough to regenerate it;
|
|
7
|
+
- manifest.json the record of the run: versions, counts, fingerprints of the files;
|
|
8
|
+
- any other file the application adds, such as the source netlist.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
from .manifest import MANIFEST
|
|
20
|
+
|
|
21
|
+
SAMPLES, WAVEFORMS, METADATA = "samples.parquet", "waveforms.npy", "metadata.json"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def assemble(stems: list[Path], out_dir: str | Path) -> tuple[pd.DataFrame, np.ndarray | None]:
|
|
25
|
+
"""Join the chunks written by a campaign into `out_dir`; returns (samples, waveforms)."""
|
|
26
|
+
out_dir = Path(out_dir)
|
|
27
|
+
df = pd.concat([pd.read_parquet(s.with_suffix(".parquet")) for s in stems], ignore_index=True)
|
|
28
|
+
df.to_parquet(out_dir / SAMPLES, index=False)
|
|
29
|
+
if not all(s.with_suffix(".npy").exists() for s in stems):
|
|
30
|
+
return df, None
|
|
31
|
+
waveforms = np.concatenate([np.load(s.with_suffix(".npy")) for s in stems])
|
|
32
|
+
np.save(out_dir / WAVEFORMS, waveforms)
|
|
33
|
+
return df, waveforms
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load_metadata(path: str | Path) -> dict:
|
|
37
|
+
"""The definition of the campaign that wrote the dataset."""
|
|
38
|
+
return json.loads((Path(path) / METADATA).read_text())
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def load_dataset(
|
|
42
|
+
path: str | Path, drop_failed: bool = True, ok_column: str = "sim_ok"
|
|
43
|
+
) -> tuple[pd.DataFrame, np.ndarray | None, dict]:
|
|
44
|
+
"""Return (samples, waveforms, manifest). Failed simulations are dropped on request."""
|
|
45
|
+
path = Path(path)
|
|
46
|
+
df = pd.read_parquet(path / SAMPLES)
|
|
47
|
+
waveforms = np.load(path / WAVEFORMS) if (path / WAVEFORMS).exists() else None
|
|
48
|
+
manifest = json.loads((path / MANIFEST).read_text())
|
|
49
|
+
if drop_failed and ok_column in df:
|
|
50
|
+
keep = df[ok_column].to_numpy()
|
|
51
|
+
df = df[keep].reset_index(drop=True)
|
|
52
|
+
waveforms = waveforms[keep] if waveforms is not None else None
|
|
53
|
+
return df, waveforms, manifest
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Experiments: deterministic sampling and parallel, resumable execution."""
|
|
2
|
+
|
|
3
|
+
from .campaign import FaultCampaign, ValidationReport, simulate_sample
|
|
4
|
+
from .engine import config_key, run_campaign, run_chunks
|
|
5
|
+
from .experiment import Experiment, ExperimentResult, Realisation, Sample, SampleResult
|
|
6
|
+
from .seeding import sample_stream
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"Experiment",
|
|
10
|
+
"ExperimentResult",
|
|
11
|
+
"FaultCampaign",
|
|
12
|
+
"Realisation",
|
|
13
|
+
"Sample",
|
|
14
|
+
"SampleResult",
|
|
15
|
+
"ValidationReport",
|
|
16
|
+
"config_key",
|
|
17
|
+
"run_campaign",
|
|
18
|
+
"run_chunks",
|
|
19
|
+
"sample_stream",
|
|
20
|
+
"simulate_sample",
|
|
21
|
+
]
|