simcon-toolkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. simcon_toolkit/__init__.py +40 -0
  2. simcon_toolkit/__main__.py +7 -0
  3. simcon_toolkit/_kit/LICENSE +202 -0
  4. simcon_toolkit/_kit/NOTICE +37 -0
  5. simcon_toolkit/_kit/assets/parts/clip_frame.stl +0 -0
  6. simcon_toolkit/_kit/assets/parts/simple_plate.stl +0 -0
  7. simcon_toolkit/_kit/packages/.ruff.toml +10 -0
  8. simcon_toolkit/_kit/packages/cadmould_cloud/__init__.py +8 -0
  9. simcon_toolkit/_kit/packages/cadmould_cloud/auth.py +681 -0
  10. simcon_toolkit/_kit/packages/cadmould_cloud/client.py +235 -0
  11. simcon_toolkit/_kit/packages/cadmould_geometry/__init__.py +5 -0
  12. simcon_toolkit/_kit/packages/cadmould_geometry/mesh.py +210 -0
  13. simcon_toolkit/_kit/packages/cadmould_geometry/stl.py +168 -0
  14. simcon_toolkit/_kit/packages/cadmould_results/__init__.py +30 -0
  15. simcon_toolkit/_kit/packages/cadmould_results/loader.py +288 -0
  16. simcon_toolkit/_kit/packages/cadmould_scoring/__init__.py +7 -0
  17. simcon_toolkit/_kit/packages/cadmould_scoring/metrics.py +519 -0
  18. simcon_toolkit/_kit/pyproject.toml +232 -0
  19. simcon_toolkit/_kit/templates/_shared/AGENTS.base.md +101 -0
  20. simcon_toolkit/_kit/templates/gate-study/.gitignore +18 -0
  21. simcon_toolkit/_kit/templates/gate-study/AGENTS.md +46 -0
  22. simcon_toolkit/_kit/templates/gate-study/GATING_STUDY_PLAYBOOK.md +219 -0
  23. simcon_toolkit/_kit/templates/gate-study/INITIAL_PROMPT.md +26 -0
  24. simcon_toolkit/_kit/templates/gate-study/README.md +137 -0
  25. simcon_toolkit/_kit/templates/gate-study/main.py +344 -0
  26. simcon_toolkit/_kit/templates/gate-study/pipeline.py +281 -0
  27. simcon_toolkit/_kit/templates/process-window/.gitignore +20 -0
  28. simcon_toolkit/_kit/templates/process-window/AGENTS.md +49 -0
  29. simcon_toolkit/_kit/templates/process-window/METHOD.md +155 -0
  30. simcon_toolkit/_kit/templates/process-window/README.md +176 -0
  31. simcon_toolkit/_kit/templates/process-window/configs/simple-plate.yaml +116 -0
  32. simcon_toolkit/_kit/templates/process-window/doe_spec.schema.md +249 -0
  33. simcon_toolkit/_kit/templates/process-window/main.py +82 -0
  34. simcon_toolkit/_kit/templates/process-window/process_window/__init__.py +5 -0
  35. simcon_toolkit/_kit/templates/process-window/process_window/centre.py +298 -0
  36. simcon_toolkit/_kit/templates/process-window/process_window/design.py +144 -0
  37. simcon_toolkit/_kit/templates/process-window/process_window/economics.py +367 -0
  38. simcon_toolkit/_kit/templates/process-window/process_window/emit.py +591 -0
  39. simcon_toolkit/_kit/templates/process-window/process_window/guardrails.py +153 -0
  40. simcon_toolkit/_kit/templates/process-window/process_window/harness.py +360 -0
  41. simcon_toolkit/_kit/templates/process-window/process_window/identity.py +92 -0
  42. simcon_toolkit/_kit/templates/process-window/process_window/inspect_part.py +184 -0
  43. simcon_toolkit/_kit/templates/process-window/process_window/kpis.py +355 -0
  44. simcon_toolkit/_kit/templates/process-window/process_window/material_card.py +163 -0
  45. simcon_toolkit/_kit/templates/process-window/process_window/probe_proxy.py +169 -0
  46. simcon_toolkit/_kit/templates/process-window/process_window/run_confirm.py +403 -0
  47. simcon_toolkit/_kit/templates/process-window/process_window/run_epsilon_floor.py +198 -0
  48. simcon_toolkit/_kit/templates/process-window/process_window/run_feedback.py +322 -0
  49. simcon_toolkit/_kit/templates/process-window/process_window/run_refine.py +279 -0
  50. simcon_toolkit/_kit/templates/process-window/process_window/run_screening.py +370 -0
  51. simcon_toolkit/_kit/templates/process-window/process_window/run_sweep.py +166 -0
  52. simcon_toolkit/_kit/templates/process-window/process_window/setup_campaign.py +312 -0
  53. simcon_toolkit/_kit/templates/process-window/process_window/surrogate.py +201 -0
  54. simcon_toolkit/_kit/templates/process-window/process_window/test_centre.py +169 -0
  55. simcon_toolkit/_kit/templates/process-window/process_window/test_design.py +113 -0
  56. simcon_toolkit/_kit/templates/process-window/process_window/test_guardrails.py +157 -0
  57. simcon_toolkit/_kit/templates/process-window/process_window/test_surrogate.py +127 -0
  58. simcon_toolkit/_kit/templates/process-window/process_window/units.py +152 -0
  59. simcon_toolkit/_kit/templates/quoting/.gitignore +24 -0
  60. simcon_toolkit/_kit/templates/quoting/AGENTS.md +58 -0
  61. simcon_toolkit/_kit/templates/quoting/INTERVIEW.md +147 -0
  62. simcon_toolkit/_kit/templates/quoting/METHOD.md +256 -0
  63. simcon_toolkit/_kit/templates/quoting/PROMPT.md +46 -0
  64. simcon_toolkit/_kit/templates/quoting/QUOTING_PLAYBOOK.md +245 -0
  65. simcon_toolkit/_kit/templates/quoting/README.md +158 -0
  66. simcon_toolkit/_kit/templates/quoting/main.py +484 -0
  67. simcon_toolkit/_kit/templates/quoting/parts/.gitkeep +0 -0
  68. simcon_toolkit/_kit/templates/quoting/quoting/__init__.py +11 -0
  69. simcon_toolkit/_kit/templates/quoting/quoting/costing.py +725 -0
  70. simcon_toolkit/_kit/templates/quoting/quoting/geometry.py +398 -0
  71. simcon_toolkit/_kit/templates/quoting/quoting/shop.py +193 -0
  72. simcon_toolkit/_kit/templates/quoting/quoting/state.py +260 -0
  73. simcon_toolkit/_kit/templates/quoting/quoting/study.py +577 -0
  74. simcon_toolkit/_kit/templates/quoting/quoting/toolkit.py +50 -0
  75. simcon_toolkit/_kit/templates/quoting/shop/README.md +43 -0
  76. simcon_toolkit/_kit/templates/quoting/shop/commercial.md +86 -0
  77. simcon_toolkit/_kit/templates/quoting/shop/lessons.md +94 -0
  78. simcon_toolkit/_kit/templates/quoting/shop/machines.md +68 -0
  79. simcon_toolkit/_kit/templates/quoting/shop/materials.md +92 -0
  80. simcon_toolkit/_kit/templates/quoting/shop/shop-profile.md +87 -0
  81. simcon_toolkit/_kit/templates/quoting/shop/tooling.md +145 -0
  82. simcon_toolkit/_kit/templates/run-one-simulation/.gitignore +16 -0
  83. simcon_toolkit/_kit/templates/run-one-simulation/AGENTS.md +41 -0
  84. simcon_toolkit/_kit/templates/run-one-simulation/README.md +133 -0
  85. simcon_toolkit/_kit/templates/run-one-simulation/main.py +216 -0
  86. simcon_toolkit/_kit/templates.toml +83 -0
  87. simcon_toolkit/choices.py +11 -0
  88. simcon_toolkit/cli.py +381 -0
  89. simcon_toolkit/generate.py +590 -0
  90. simcon_toolkit/instructions.py +152 -0
  91. simcon_toolkit/manifest.py +86 -0
  92. simcon_toolkit/project.py +356 -0
  93. simcon_toolkit/wizard.py +160 -0
  94. simcon_toolkit-0.1.0.dist-info/METADATA +48 -0
  95. simcon_toolkit-0.1.0.dist-info/RECORD +97 -0
  96. simcon_toolkit-0.1.0.dist-info/WHEEL +4 -0
  97. simcon_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,153 @@
1
+ """Guardrail enforcement. These raise; they are not warnings.
2
+
3
+ The brief asks for a lint that **fails the run** if an absolute pressure threshold appears in a
4
+ screening criterion. The reason is specific and quantitative: on multi-gate parts the AI solver
5
+ under-predicts absolute cavity pressure by roughly a factor of three (~0.3x measured on an
6
+ 8-gate part). This campaign models 1 of a real 32 cavities. A criterion of the form
7
+ `p80 < 250 bar` is therefore comparing a number known to be wrong against a number believed to
8
+ be right — it will happily pass or fail points for reasons that have nothing to do with the
9
+ mould.
10
+
11
+ Ranks, ratios and comparisons survive that bias because a monotone distortion preserves
12
+ ordering. So the rule is not "be careful with pressure", it is "pressure may only ever be
13
+ *compared*, never *thresholded*", and a comment cannot enforce that.
14
+
15
+ Note the deliberate exception: a **machine limit** (1500 bar, 1000 kN) is a constraint, not a
16
+ criterion. It may annotate the final recommendation, where it is tagged `absolute: true` and
17
+ carries the caveats. `check_absolute_claim_has_caveats` enforces that side.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+
24
+ # Units that carry an absolute pressure scale. Dimensionless ratios and rank fractions do not.
25
+ _PRESSURE_UNITS = r"(?:bar|mbar|kpa|mpa|psi|pa)\b"
26
+
27
+ # A number next to a pressure unit, e.g. "250 bar", "1.5e7 Pa", "p80<250bar".
28
+ _ABS_PRESSURE = re.compile(rf"\d+(?:\.\d+)?(?:[eE][+-]?\d+)?\s*{_PRESSURE_UNITS}", re.I)
29
+
30
+ # A comparison against a bare number involving a pressure-named quantity, e.g. "p80 <= 250",
31
+ # "max_pressure_bar > 300". Catches the case where the unit is in the variable name instead.
32
+ _PRESSURE_NAME = r"(?:p80|p_?max|peak_?pressure|max_?pressure|cavity_?pressure|pressure)"
33
+ _NAMED_CMP = re.compile(rf"{_PRESSURE_NAME}\w*\s*(?:<=|>=|<|>|==)\s*[-+]?\d", re.I)
34
+
35
+ # Forms that are legitimate because they are relative, not absolute.
36
+ _RELATIVE_OK = re.compile(
37
+ r"_rank\b|_ratio\b|_pct\b|_percentile\b|_quantile\b|_norm\b|_rel\b|_share\b"
38
+ r"|rank\(|percentile\(|quantile\(|_z\b",
39
+ re.I,
40
+ )
41
+
42
+ # One pressure divided by another is dimensionless, so it is a ratio and permitted — e.g.
43
+ # `p80 / p80_centre < 1.15`. Dividing by an absolute instead (`p80 / 250 bar`) is still caught
44
+ # by _ABS_PRESSURE, so this cannot be used to smuggle a threshold in.
45
+ _PRESSURE_RATIO = re.compile(rf"{_PRESSURE_NAME}\w*\s*/\s*{_PRESSURE_NAME}\w*", re.I)
46
+
47
+
48
+ class GuardrailViolation(AssertionError):
49
+ """A criterion or claim broke a guardrail. Fails the run by design."""
50
+
51
+
52
+ def check_criterion(name: str, expression: str) -> None:
53
+ """Raise if ``expression`` thresholds an absolute pressure.
54
+
55
+ Applied to every entry of `acceptance_criteria` before it can filter anything.
56
+ """
57
+ expr = str(expression)
58
+ relative = bool(_RELATIVE_OK.search(expr)) or bool(_PRESSURE_RATIO.search(expr))
59
+
60
+ if _ABS_PRESSURE.search(expr):
61
+ raise GuardrailViolation(
62
+ f"criterion '{name}' contains an absolute pressure threshold: {expr!r}\n"
63
+ " Guardrail 1: the AI solver under-predicts absolute pressure on multi-gate parts "
64
+ "(~0.3x on 8 gates); this campaign models 1 of 32 cavities with no runner system.\n"
65
+ " Use a rank, ratio or comparison instead, e.g. 'p80_rank <= 0.70'."
66
+ )
67
+ if _NAMED_CMP.search(expr) and not relative:
68
+ raise GuardrailViolation(
69
+ f"criterion '{name}' compares a pressure quantity to a bare number: {expr!r}\n"
70
+ " Guardrail 1: pressure may be ordered, never thresholded. If this really is a "
71
+ "rank or ratio, name it so (…_rank, …_ratio) — the name is the audit trail."
72
+ )
73
+
74
+
75
+ def check_criteria(criteria: list[dict]) -> None:
76
+ """Lint a whole `acceptance_criteria` list. Raises on the first violation."""
77
+ for c in criteria:
78
+ check_criterion(c.get("name", "<unnamed>"), c.get("expression", ""))
79
+ kind = c.get("kind")
80
+ if kind not in ("rank", "ratio", "comparison"):
81
+ raise GuardrailViolation(
82
+ f"criterion '{c.get('name')}' has kind={kind!r}; guardrail 1 allows only "
83
+ "'rank', 'ratio' or 'comparison'."
84
+ )
85
+ if c.get("absolute"):
86
+ raise GuardrailViolation(
87
+ f"criterion '{c.get('name')}' is marked absolute:true. Absolute values may "
88
+ "annotate the final recommendation but may never screen a point."
89
+ )
90
+
91
+
92
+ def check_absolute_claim_has_caveats(key: str, block: dict) -> None:
93
+ """The other side of the rule: an absolute-valued claim must carry its caveats.
94
+
95
+ Machine limits and any reported pressure in bar are allowed *as annotations*, but only if
96
+ they state what they rest on. An uncaveated absolute number is how "147 bar is comfortable"
97
+ escapes into a slide deck without its 1-of-32-cavities asterisk.
98
+ """
99
+ if not block.get("absolute"):
100
+ return
101
+ if not block.get("caveats"):
102
+ raise GuardrailViolation(
103
+ f"'{key}' is marked absolute:true but carries no caveats. Required: the "
104
+ "multi-gate pressure under-prediction, the 1-of-32-cavities limitation, and the "
105
+ "absence of any runner system."
106
+ )
107
+
108
+
109
+ def check_noise_floor_derived(floor: dict) -> None:
110
+ """Guardrail 2: the floor must come from this part, not from the reference campaign."""
111
+ if not floor:
112
+ raise GuardrailViolation("no noise floor recorded; guardrail 2 requires one per KPI.")
113
+ for kpi, d in floor.items():
114
+ if d.get("basis") != "derived":
115
+ raise GuardrailViolation(
116
+ f"noise floor for '{kpi}' has basis={d.get('basis')!r}, expected 'derived'. "
117
+ "The reference part's +/-1.2 bar / 2 bar figures are part-specific and must "
118
+ "not be imported."
119
+ )
120
+ if not d.get("n_runs"):
121
+ raise GuardrailViolation(f"noise floor for '{kpi}' records no run count.")
122
+
123
+
124
+ def check_forbidden_kpi(criteria: list[dict]) -> None:
125
+ """Fill-race metrics are measured not to transfer (Spearman 0.10-0.23).
126
+
127
+ Allowed as a labelled anomaly detector, never as a screening criterion.
128
+ """
129
+ for c in criteria:
130
+ blob = f"{c.get('name', '')} {c.get('kpi', '')} {c.get('expression', '')}".lower()
131
+ if "race" in blob or "fill_time_diff" in blob or "filltimediff" in blob:
132
+ raise GuardrailViolation(
133
+ f"criterion '{c.get('name')}' screens on a fill-race metric. Measured not to "
134
+ "transfer (Spearman 0.10-0.23); permitted only as a labelled anomaly detector."
135
+ )
136
+
137
+
138
+ def lint_spec(spec: dict) -> list[str]:
139
+ """Run every guardrail over a doe_spec dict. Returns the passed checks; raises on failure."""
140
+ passed = []
141
+ criteria = spec.get("acceptance_criteria", []) or []
142
+ check_criteria(criteria)
143
+ passed.append(f"acceptance_criteria: {len(criteria)} checked, none absolute")
144
+ check_forbidden_kpi(criteria)
145
+ passed.append("no fill-race metric used as a screening criterion")
146
+ check_noise_floor_derived((spec.get("noise_floor") or {}).get("per_kpi", {}))
147
+ passed.append("noise floor derived per KPI, not imported")
148
+ for key in ("machine_limits",):
149
+ blk = (spec.get("provenance") or {}).get(key)
150
+ if isinstance(blk, dict):
151
+ check_absolute_claim_has_caveats(key, blk)
152
+ passed.append(f"{key}: absolute claim carries caveats")
153
+ return passed
@@ -0,0 +1,360 @@
1
+ """Campaign harness: submit, score, discard, resume.
2
+
3
+ One object that every stage (epsilon floor, G1 sweep, G3 refinement) drives. What it
4
+ guarantees, and why each one is load-bearing:
5
+
6
+ **Resume, not restart.** Every run's identity is a deterministic hash of its inputs, so
7
+ "already done" is a file-existence check. Re-running the command picks up exactly where it
8
+ stopped. Assume the session dies mid-sweep — over ~1 000 runs it will.
9
+
10
+ **Never repeat a configuration.** The solver is deterministic: an identical request returns an
11
+ identical result. So a repeat buys literally nothing — not confidence, not an error bar. The
12
+ run id *is* the config hash, which makes duplicate submission structurally impossible rather
13
+ than merely discouraged.
14
+
15
+ **Score then discard.** Results are ~29 MB each; 1 024 of them is 30 GB, and the project folder
16
+ is inside OneDrive. Each result is downloaded to a scratch path **outside** OneDrive, scored to
17
+ a ~2 KB json, and deleted — unless it is flagged worth keeping (boundary, anomaly, anchor), in
18
+ which case it is moved into `cache/runs/` deliberately.
19
+
20
+ **Bounded concurrency with backoff.** The endpoint sheds load; 429/5xx are expected, not
21
+ exceptional. One warm-up request goes out before the fan-out, because a cold model 504s.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import concurrent.futures as cf
27
+ import hashlib
28
+ import json
29
+ import os
30
+ import shutil
31
+ import tempfile
32
+ import threading
33
+ import time
34
+ from dataclasses import asdict, dataclass, field
35
+ from datetime import UTC, datetime
36
+ from pathlib import Path
37
+
38
+ import httpx
39
+
40
+ from . import kpis as K
41
+ from .units import Process
42
+
43
+ RETRY_STATUS = (429, 500, 502, 503, 504)
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class RunSpec:
48
+ """One process point. ``run_id`` is derived, so identity == configuration."""
49
+
50
+ melt_C: float
51
+ wall_C: float
52
+ flow_cm3_s: float
53
+ stage: str
54
+ label: str = ""
55
+ keep_h5: bool = False
56
+ meta: dict = field(default_factory=dict)
57
+ # Hidden-deviation overrides for the feedback loop: what is ACTUALLY submitted may differ
58
+ # from what the DOE side believes it asked for. Both are part of the run identity, or a
59
+ # deviated run would collide with its own baseline.
60
+ material_id_override: str | None = None
61
+ gates_mm_override: tuple | None = None
62
+
63
+ def process(self, volume_cm3: float) -> Process:
64
+ return Process(melt_C=self.melt_C, wall_C=self.wall_C, flow_cm3_s=self.flow_cm3_s, volume_cm3=volume_cm3)
65
+
66
+ def run_id(self, *, geometry_id: str, model_version: str, num_timesteps: int, material_id: str) -> str:
67
+ """Deterministic id over everything that can change the record.
68
+
69
+ Includes geometry and solver settings, not just the process point: the same melt/wall/flow
70
+ on a different mesh or a different model version is a *different* run, and collapsing them
71
+ would silently reuse a stale result.
72
+
73
+ It also includes ``kpis.KPI_VERSION``, because a record is (inputs -> KPIs). Redefining a
74
+ KPI changes what the record *means*, so old records must be invalidated rather than reused;
75
+ without this, adding a KPI would leave every existing run permanently missing it while the
76
+ resume logic reported them as done.
77
+
78
+ ``material_id`` is the campaign's own material, not just the per-run override. Without it,
79
+ changing the material in the config left every run id unchanged, so the campaign resumed
80
+ onto the previous material's results and reported them as the new material's.
81
+ """
82
+ canon = (
83
+ f"{self.melt_C:.6f}|{self.wall_C:.6f}|{self.flow_cm3_s:.6f}"
84
+ f"|{geometry_id}|{model_version}|{num_timesteps}|{K.KPI_VERSION}"
85
+ f"|{material_id}"
86
+ f"|{self.material_id_override or '-'}|{self.gates_mm_override or '-'}"
87
+ )
88
+ return hashlib.sha1(canon.encode()).hexdigest()[:12]
89
+
90
+
91
+ @dataclass
92
+ class RunRecord:
93
+ run_id: str
94
+ spec: dict
95
+ simulation_id: str = ""
96
+ kpis: dict | None = None
97
+ h5_kept: str | None = None
98
+ error: str = ""
99
+ wall_clock_s: float = 0.0
100
+ submitted_at: str = ""
101
+ # Records from superseded KPI definitions stay on disk as history, so every record must say
102
+ # which definition produced it — otherwise a reader silently mixes incompatible generations.
103
+ kpi_version: str = ""
104
+
105
+ @property
106
+ def ok(self) -> bool:
107
+ return self.kpis is not None and not self.error
108
+
109
+
110
+ def scratch_root() -> Path:
111
+ """A download area OUTSIDE OneDrive.
112
+
113
+ Downloading 30 GB into a synced folder only to delete it again would push every byte
114
+ through OneDrive's uploader. The kept results are moved into the project deliberately;
115
+ the transient ones never touch it.
116
+ """
117
+ base = os.environ.get("LOCALAPPDATA") or tempfile.gettempdir()
118
+ p = Path(base) / "doe_runs_scratch"
119
+ p.mkdir(parents=True, exist_ok=True)
120
+ return p
121
+
122
+
123
+ class Campaign:
124
+ """One part + one material + one cloud project; many staged, resumable batches."""
125
+
126
+ def __init__(
127
+ self,
128
+ *,
129
+ api,
130
+ project_id: str,
131
+ geometry_id: str,
132
+ material_id: str,
133
+ volume_cm3: float,
134
+ card,
135
+ long_axis: int,
136
+ gates_mm: list[list[float]],
137
+ runs_dir: Path,
138
+ model_version: str,
139
+ num_timesteps: int,
140
+ htc_W_m2K: float,
141
+ switchover_phi: tuple[float, ...],
142
+ logger=print,
143
+ ):
144
+ self.api = api
145
+ self.project_id = project_id
146
+ self.geometry_id = geometry_id
147
+ self.material_id = material_id
148
+ self.volume_cm3 = volume_cm3
149
+ self.card = card
150
+ self.long_axis = long_axis
151
+ self.gates_mm = gates_mm
152
+ self.runs_dir = Path(runs_dir)
153
+ self.model_version = model_version
154
+ self.num_timesteps = num_timesteps
155
+ self.htc_W_m2K = htc_W_m2K
156
+ self.switchover_phi = switchover_phi
157
+ self.logger = logger
158
+ self._lock = threading.Lock()
159
+ self._warm = False
160
+ self._groups: dict[str, str] = {}
161
+
162
+ # -- persistence ---------------------------------------------------------
163
+ def record_path(self, stage: str, run_id: str) -> Path:
164
+ return self.runs_dir / stage / f"{run_id}.json"
165
+
166
+ def load_record(self, stage: str, run_id: str) -> RunRecord | None:
167
+ p = self.record_path(stage, run_id)
168
+ if not p.exists():
169
+ return None
170
+ try:
171
+ d = json.loads(p.read_text())
172
+ except (OSError, ValueError):
173
+ return None
174
+ # A record that failed is not "done" — retry it on the next pass.
175
+ if d.get("kpis") is None:
176
+ return None
177
+ return RunRecord(**d)
178
+
179
+ def save_record(self, stage: str, rec: RunRecord) -> None:
180
+ p = self.record_path(stage, rec.run_id)
181
+ p.parent.mkdir(parents=True, exist_ok=True)
182
+ tmp = p.with_suffix(".tmp")
183
+ tmp.write_text(json.dumps(asdict(rec), indent=2, default=str))
184
+ tmp.replace(p) # atomic: a killed process never leaves a half-written record
185
+
186
+ # -- cloud ---------------------------------------------------------------
187
+ def group_for(self, stage: str, description: str = "") -> str:
188
+ with self._lock:
189
+ if stage not in self._groups:
190
+ grp = self.api.create_group(
191
+ self.project_id, stage, description=description or f"{stage} (frontloaded DOE POC)"
192
+ )
193
+ self._groups[stage] = grp["id"]
194
+ return self._groups[stage]
195
+
196
+ def note(self, text: str) -> None:
197
+ """Append to the cloud-side decision log; never let it break a run."""
198
+ try:
199
+ self.api.add_note(self.project_id, text)
200
+ except Exception as exc:
201
+ self.logger(f" (cloud note skipped: {type(exc).__name__}: {exc})")
202
+
203
+ def _submit_with_retry(self, spec: RunSpec, group_id: str, *, attempts: int = 4, backoff: float = 2.0) -> dict:
204
+ proc = spec.process(self.volume_cm3)
205
+ last: Exception | None = None
206
+ for i in range(attempts + 1):
207
+ try:
208
+ return self.api.run_filling(
209
+ geometry_id=self.geometry_id,
210
+ material_id=spec.material_id_override or self.material_id,
211
+ gate_positions_mm=(
212
+ [list(g) for g in spec.gates_mm_override] if spec.gates_mm_override else self.gates_mm
213
+ ),
214
+ model_version=self.model_version,
215
+ num_timesteps=self.num_timesteps,
216
+ htc_W_m2K=self.htc_W_m2K,
217
+ label=spec.label or proc.label(),
218
+ project_id=self.project_id,
219
+ group_id=group_id,
220
+ solver_type="ai",
221
+ **proc.api_kwargs(),
222
+ )
223
+ except httpx.HTTPStatusError as exc:
224
+ last = exc
225
+ if exc.response.status_code not in RETRY_STATUS:
226
+ raise
227
+ except (httpx.TimeoutException, httpx.TransportError) as exc:
228
+ last = exc
229
+ if i < attempts:
230
+ time.sleep(backoff * (i + 1))
231
+ raise last # type: ignore[misc]
232
+
233
+ def warm_up(self, spec: RunSpec) -> None:
234
+ """One lone request before fanning out — a cold model 504s under a burst."""
235
+ if self._warm:
236
+ return
237
+ self.logger(" warming the model with a single request ...")
238
+ try:
239
+ self._run_one(spec, self.group_for(spec.stage))
240
+ except Exception as exc:
241
+ self.logger(f" warm-up failed ({type(exc).__name__}); continuing anyway")
242
+ self._warm = True
243
+
244
+ # -- one run -------------------------------------------------------------
245
+ def _run_one(self, spec: RunSpec, group_id: str) -> RunRecord:
246
+ run_id = spec.run_id(
247
+ geometry_id=self.geometry_id,
248
+ model_version=self.model_version,
249
+ num_timesteps=self.num_timesteps,
250
+ material_id=self.material_id,
251
+ )
252
+ rec = RunRecord(
253
+ run_id=run_id,
254
+ spec=asdict(spec),
255
+ submitted_at=datetime.now(UTC).isoformat(),
256
+ kpi_version=K.KPI_VERSION,
257
+ )
258
+ h5 = scratch_root() / f"{run_id}.h5"
259
+ try:
260
+ t0 = time.time()
261
+ res = self._submit_with_retry(spec, group_id)
262
+ rec.simulation_id = res["simulation_id"]
263
+ self.api.download_to(res["result"]["download"]["url"], h5)
264
+ rec.wall_clock_s = round(time.time() - t0, 2)
265
+
266
+ proc = spec.process(self.volume_cm3)
267
+ k = K.extract(
268
+ h5,
269
+ melt_K=proc.melt_K,
270
+ wall_K=proc.wall_K,
271
+ no_flow_K=self.card.no_flow_K,
272
+ switchover_phi=self.switchover_phi,
273
+ long_axis=self.long_axis,
274
+ )
275
+ rec.kpis = k.as_dict()
276
+
277
+ if spec.keep_h5:
278
+ dest = self.runs_dir / spec.stage / f"{run_id}.h5"
279
+ dest.parent.mkdir(parents=True, exist_ok=True)
280
+ shutil.move(str(h5), str(dest))
281
+ rec.h5_kept = str(dest.relative_to(self.runs_dir))
282
+ except Exception as exc:
283
+ rec.error = f"{type(exc).__name__}: {exc}"
284
+ finally:
285
+ h5.unlink(missing_ok=True) # score-then-discard
286
+ self.save_record(spec.stage, rec)
287
+ return rec
288
+
289
+ # -- a batch -------------------------------------------------------------
290
+ def run_batch(
291
+ self, specs: list[RunSpec], *, stage: str, max_workers: int = 4, description: str = ""
292
+ ) -> list[RunRecord]:
293
+ """Run every spec not already recorded. Returns records for the whole batch."""
294
+ done: list[RunRecord] = []
295
+ todo: list[RunSpec] = []
296
+ for s in specs:
297
+ rid = s.run_id(
298
+ geometry_id=self.geometry_id,
299
+ model_version=self.model_version,
300
+ num_timesteps=self.num_timesteps,
301
+ material_id=self.material_id,
302
+ )
303
+ existing = self.load_record(stage, rid)
304
+ if existing:
305
+ done.append(existing)
306
+ else:
307
+ todo.append(s)
308
+
309
+ self.logger(f"\n stage '{stage}': {len(specs)} specs -> {len(done)} already done, {len(todo)} to run")
310
+ if not todo:
311
+ return done
312
+
313
+ group_id = self.group_for(stage, description)
314
+ self.warm_up(todo[0])
315
+ # warm_up ran todo[0] and saved its record; skip it in the fan-out.
316
+ remaining = [
317
+ s
318
+ for s in todo
319
+ if not self.load_record(
320
+ stage,
321
+ s.run_id(
322
+ geometry_id=self.geometry_id,
323
+ model_version=self.model_version,
324
+ num_timesteps=self.num_timesteps,
325
+ material_id=self.material_id,
326
+ ),
327
+ )
328
+ ]
329
+
330
+ results: list[RunRecord] = []
331
+ t0 = time.time()
332
+ with cf.ThreadPoolExecutor(max_workers=max(1, min(max_workers, len(remaining) or 1))) as pool:
333
+ futs = {pool.submit(self._run_one, s, group_id): s for s in remaining}
334
+ for n, fut in enumerate(cf.as_completed(futs), 1):
335
+ rec = fut.result()
336
+ results.append(rec)
337
+ if rec.ok:
338
+ self.logger(
339
+ f" [{n}/{len(remaining)}] {rec.spec['label'] or rec.run_id} "
340
+ f"p80={rec.kpis['p80_bar']}bar "
341
+ f"noflow={rec.kpis['no_flow_margin']} ({rec.wall_clock_s}s)"
342
+ )
343
+ else:
344
+ self.logger(f" [{n}/{len(remaining)}] {rec.run_id} FAILED {rec.error}")
345
+ ok = sum(1 for r in results if r.ok)
346
+ self.logger(f" stage '{stage}': {ok}/{len(remaining)} succeeded in {time.time() - t0:.0f}s")
347
+
348
+ # Re-read everything so the caller gets one consistent view including the warm-up run.
349
+ out: list[RunRecord] = []
350
+ for s in specs:
351
+ rid = s.run_id(
352
+ geometry_id=self.geometry_id,
353
+ model_version=self.model_version,
354
+ num_timesteps=self.num_timesteps,
355
+ material_id=self.material_id,
356
+ )
357
+ r = self.load_record(stage, rid)
358
+ if r:
359
+ out.append(r)
360
+ return out
@@ -0,0 +1,92 @@
1
+ """Which campaign a part belongs to.
2
+
3
+ Deliberately free of the licensed SDK and of everything else heavy, for two reasons: it
4
+ is the one thing every stage needs before it can find its own folder, and it is the part
5
+ of the campaign whose failures are silent, so it has to be testable in the lane that
6
+ gates merges.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ from functools import cache
13
+ from pathlib import Path
14
+
15
+
16
+ def slug(name: str) -> str:
17
+ return "".join(c if c.isalnum() else "_" for c in str(name).lower()).strip("_")
18
+
19
+
20
+ @cache
21
+ def _digest(part: Path, mtime_ns: int, size: int) -> str:
22
+ """Hash a part's contents. Cached per process; mtime and size keep the cache honest."""
23
+ h = hashlib.sha256()
24
+ with part.open("rb") as fh:
25
+ for chunk in iter(lambda: fh.read(1 << 20), b""):
26
+ h.update(chunk)
27
+ return h.hexdigest()[:12]
28
+
29
+
30
+ def part_key(part: Path, scale: float = 1.0) -> str:
31
+ """Cache identity for a part: its name, plus enough of its content to be honest.
32
+
33
+ Neither the file name nor the `name:` in the config is an identity. Two customers'
34
+ parts can both be bracket.stl, and re-exporting a part changes its bytes but not its
35
+ path — the ordinary working loop here. Keyed on either of those, a campaign reuses the
36
+ previous part's mesh and uploaded geometry and reports a process window for a part
37
+ that no longer exists, with nothing anywhere to show that it happened.
38
+
39
+ The scale rides along because it changes what the mesh means without touching a byte
40
+ of the file: meshing at 0.1 and computing a volume at 1.0 disagree silently. It is
41
+ omitted at 1.0, which is nearly every campaign, so folder names stay readable.
42
+ """
43
+ try:
44
+ st = part.stat()
45
+ except FileNotFoundError:
46
+ raise SystemExit(
47
+ f"part not found: {part}\n"
48
+ "A campaign is filed under the part it describes, so the file the config's "
49
+ "`part.stl` names has to be present. Point `part.stl` at your geometry, or "
50
+ "restore it."
51
+ ) from None
52
+ suffix = "" if scale == 1.0 else f"-x{scale:g}"
53
+ return f"{slug(part.stem)}-{_digest(part, st.st_mtime_ns, st.st_size)}{suffix}"
54
+
55
+
56
+ def find_existing_campaign(runs_root: Path, part: Path) -> Path | None:
57
+ """The campaign folder for a part whose file is gone, when there is exactly one.
58
+
59
+ Re-emitting a dossier from a kept `cache/` is a reasonable thing to want on a machine
60
+ that no longer has the geometry. One match is unambiguous; several are not, and
61
+ guessing between two versions of the same part is the failure this module exists to
62
+ prevent, so the caller is told to say which.
63
+ """
64
+ if not runs_root.is_dir():
65
+ return None
66
+ matches = sorted(d for d in runs_root.glob(f"{slug(part.stem)}-*") if d.is_dir())
67
+ if len(matches) == 1:
68
+ return matches[0]
69
+ if len(matches) > 1:
70
+ names = "\n ".join(d.name for d in matches)
71
+ raise SystemExit(
72
+ f"part not found: {part}\n"
73
+ f"Several saved campaigns match that name, so which one you mean is a guess:\n {names}\n"
74
+ "Restore the geometry, or point `part.stl` at the version you want."
75
+ )
76
+ return None
77
+
78
+
79
+ def foreign_campaigns(runs_root: Path, default_key: str | None) -> list[str]:
80
+ """Saved campaigns that the bundled sample's config does not account for.
81
+
82
+ A stage run with no config falls back to the sample's. That is convenient and, once a
83
+ second campaign exists on disk, it is also a way to score the wrong part in silence:
84
+ the engineer runs one stage against their own config, then a bare one, and the bare
85
+ one reads and writes under the sample's key while appearing to continue their work.
86
+
87
+ An unknown ``default_key`` — the sample itself is missing, so its key cannot be
88
+ computed — makes every campaign foreign. Refusing costs a retype; guessing costs a run.
89
+ """
90
+ if not runs_root.is_dir():
91
+ return []
92
+ return sorted(d.name for d in runs_root.iterdir() if d.is_dir() and d.name != default_key)