methodlm 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
methodlm.py ADDED
@@ -0,0 +1,839 @@
1
+ #!/usr/bin/env python3
2
+ """MethodLM -- the method, in a language model, kept honest by a ledger.
3
+
4
+ Point it at data; it computes on it (ternary two-timescale readout) and reasons about
5
+ it (gbranaa-hue method), pre-registering every test.
6
+
7
+ python methodlm.py --demo interventional demo world (hidden
8
+ confound + answer key at the end)
9
+ python methodlm.py --diabetes real clinical data (442 patients,
10
+ sklearn load_diabetes, raw units)
11
+ python methodlm.py --csv F --target COL any recorded CSV (observational)
12
+
13
+ Pipeline (identical in all modes):
14
+ COMPUTE a tritkit TwoTimescaleLinear ternary readout learns target from the other
15
+ columns -> NMSE + structural evidence share + gate selection.
16
+ REASON the method copilot (Qwen-3B + gbranaa-hue method) investigates with tools,
17
+ pre-registers every test, writes an honest ledger.
18
+
19
+ Tools by mode: CORR (all) | ATTR (all) | RUN true intervention (demo world only)
20
+ STRAT observational conditioning (recorded data): corr(x,target)
21
+ inside quartile bands of z -- the honest substitute for clamping
22
+ when you cannot rerun the world.
23
+ """
24
+ import argparse, os, re, sys, subprocess, textwrap, time
25
+ import numpy as np
26
+
27
+ HERE = os.path.dirname(os.path.abspath(__file__))
28
+ _tk = os.environ.get("METHODLM_TRITKIT") # optional: path to the tritkit pkg (ternary 2nd witness)
29
+ if _tk:
30
+ sys.path.insert(0, _tk)
31
+ import methodlm_models
32
+ rng = np.random.default_rng(11)
33
+
34
+ BACKEND = None # the model driving the reasoning; set by main() or lazily to local
35
+ def backend():
36
+ global BACKEND
37
+ if BACKEND is None:
38
+ BACKEND = methodlm_models.get_model(os.environ.get("METHODLM_MODEL", "local"), HERE)
39
+ return BACKEND
40
+
41
+ # ---------------- data sources ----------------
42
+ def demo_world(n=2000, clamp=None):
43
+ clamp = clamp or {}
44
+ season = np.cumsum(rng.normal(0, 0.15, n)); season -= season.mean()
45
+ d = {"temperature": 25 + 6 * np.tanh(season) + rng.normal(0, 0.8, n),
46
+ "humidity": 50 + 15 * np.tanh(season) + rng.normal(0, 2.5, n),
47
+ "vibration": rng.normal(0, 1, n)}
48
+ for k, v in clamp.items():
49
+ if k in d: d[k] = np.full(n, float(v))
50
+ d["error"] = np.clip(0.5 + 0.08 * (d["humidity"] - 50) + rng.normal(0, 0.35, n), 0, None)
51
+ return d
52
+
53
+ def load_diabetes():
54
+ from sklearn.datasets import load_diabetes as ld
55
+ raw = ld(scaled=False)
56
+ names = ["age", "sex", "bmi", "bp", "tc", "ldl", "hdl", "tch", "ltg", "glu"]
57
+ d = {n: raw.data[:, i].astype(float) for i, n in enumerate(names)}
58
+ d["progression"] = raw.target.astype(float)
59
+ return d
60
+
61
+ def load_csv(path, target):
62
+ import csv as _csv
63
+ with open(path, newline="", encoding="utf-8-sig") as f:
64
+ rows = list(_csv.DictReader(f))
65
+ cols = {}
66
+ for k in rows[0]:
67
+ try:
68
+ cols[k.strip()] = np.array([float(r[k]) for r in rows])
69
+ except ValueError:
70
+ pass # skip non-numeric columns
71
+ assert target in cols, f"target '{target}' not among numeric columns {list(cols)}"
72
+ return cols
73
+
74
+ # ---------------- COMPUTE: ternary two-timescale readout ----------------
75
+ def train_readout(data, target):
76
+ import torch
77
+ import torch.nn.functional as F
78
+ from tritkit.twotimescale import TwoTimescaleLinear
79
+ torch.manual_seed(0)
80
+ names = [k for k in data if k != target]
81
+ X = np.column_stack([data[k] for k in names])
82
+ X = (X - X.mean(0)) / (X.std(0) + 1e-9)
83
+ Yv = data[target]; Yn = (Yv - Yv.mean()) / Yv.std()
84
+ Xt, Yt = torch.tensor(X, dtype=torch.float32), torch.tensor(Yn, dtype=torch.float32)
85
+ n = len(Yn)
86
+ lyr = TwoTimescaleLinear(len(names), 1, density=min(0.5, 3 / len(names) + 0.15),
87
+ bias=True, evidence_beta=0.97)
88
+ opt = torch.optim.SGD(lyr.parameters(), lr=0.02)
89
+ e2 = []
90
+ for ep in range(10):
91
+ perm = torch.randperm(n)
92
+ for t in range(0, n - 16, 16):
93
+ idx = perm[t:t + 16]
94
+ loss = F.mse_loss(lyr(Xt[idx]), Yt[idx, None])
95
+ opt.zero_grad(); loss.backward(); opt.step()
96
+ if t % 320 == 0: lyr.step_gate()
97
+ if ep == 9: e2.append(loss.item())
98
+ ev = lyr.evidence[0].detach().numpy(); share = ev / ev.sum()
99
+ order = np.argsort(-share)
100
+ return (float(np.mean(e2)),
101
+ {names[i]: float(share[i]) for i in order},
102
+ [names[i] for i in range(len(names)) if lyr.G[0, i] > 0])
103
+
104
+ # ---------------- tools ----------------
105
+ _DOWHY_OK = None
106
+ def _dowhy_available():
107
+ """Same lazy-singleton optional-dependency pattern as RECALL's
108
+ _vault_engine() below: tried-and-failed is cached (True/False), not
109
+ retried every call."""
110
+ global _DOWHY_OK
111
+ if _DOWHY_OK is None:
112
+ try:
113
+ import dowhy # noqa: F401
114
+ _DOWHY_OK = True
115
+ except ImportError:
116
+ _DOWHY_OK = False
117
+ return _DOWHY_OK
118
+
119
+ def make_tools(data, target, interventional):
120
+ def corr(a, b):
121
+ if a not in data or b not in data:
122
+ return f"unknown column(s); columns are {list(data)}"
123
+ r = float(np.corrcoef(data[a], data[b])[0, 1])
124
+ return f"corr({a},{b}) = {r:+.2f} over {len(data[a])} samples."
125
+
126
+ def run(vary, clamp):
127
+ if not interventional:
128
+ return "RUN unavailable: this is recorded data, not a rerunnable system. Use STRAT."
129
+ d = demo_world(400, clamp=clamp)
130
+ r = float(np.corrcoef(d[vary], d[target])[0, 1]) if vary in d else float("nan")
131
+ cl = ", ".join(f"{k}={v}" for k, v in clamp.items()) or "nothing"
132
+ return (f"Controlled run: 400 fresh trials varying {vary}, clamping {cl}. "
133
+ f"corr({vary},{target}) = {r:+.2f}; mean {target} {d[target].mean():.2f}.")
134
+
135
+ def strat(x, z):
136
+ if x not in data or z not in data:
137
+ return f"unknown column(s); columns are {list(data)}"
138
+ raw = float(np.corrcoef(data[x], data[target])[0, 1])
139
+ qs = np.quantile(data[z], [0, .25, .5, .75, 1.0])
140
+ rs = []
141
+ for i in range(4):
142
+ m = (data[z] >= qs[i]) & (data[z] <= qs[i + 1])
143
+ if m.sum() > 20 and np.std(data[x][m]) > 0:
144
+ rs.append(float(np.corrcoef(data[x][m], data[target][m])[0, 1]))
145
+ within = float(np.mean(rs)) if rs else float("nan")
146
+ return (f"Stratified: raw corr({x},{target}) = {raw:+.2f}; inside quartile bands of {z} "
147
+ f"it is {['%+.2f' % r for r in rs]} (mean {within:+.2f}). "
148
+ f"If the within-band mean collapses, {x}'s link runs through {z}.")
149
+
150
+ def adjust(x, zs):
151
+ """Backdoor adjustment (multiple regression) + Cinelli-Hazlett sensitivity, WITH a
152
+ collider/mediator bias audit. Adjusting for a MEDIATOR (on the X->target path) or a
153
+ COLLIDER (a common effect of X and target) INTRODUCES bias -- the 'Table 2 fallacy' --
154
+ so 'control for everything' is wrong for observational/causal data. Data alone cannot
155
+ prove a variable's role (a confounder and a mediator are observationally identical);
156
+ this flags the one danger that IS detectable (a collider: conditioning opens a path
157
+ and RAISES the X-target association) and defers the rest to the DAG / time-order."""
158
+ if x not in data:
159
+ return f"unknown column '{x}'; columns are {list(data)}"
160
+ others = [c for c in data if c not in (x, target)]
161
+ zs = [z for z in zs if z in data and z not in (x, target)]
162
+ n = len(data[target])
163
+
164
+ def zsc(a):
165
+ a = np.asarray(a, float); return (a - a.mean()) / (a.std() + 1e-9)
166
+ y = zsc(data[target])
167
+
168
+ def fit(cond): # partial corr, t, RV of x | cond
169
+ X = np.column_stack([zsc(data[c]) for c in [x] + cond] + [np.ones(n)])
170
+ beta, *_ = np.linalg.lstsq(X, y, rcond=None)
171
+ resid = y - X @ beta; dof = n - X.shape[1]
172
+ se = np.sqrt(((resid ** 2).sum() / max(dof, 1)) * np.diag(np.linalg.pinv(X.T @ X)))
173
+ t = float(beta[0] / (se[0] + 1e-12))
174
+ partial = t / np.sqrt(t * t + dof) if dof > 0 else float("nan")
175
+ f = abs(t) / np.sqrt(max(dof, 1))
176
+ return partial, t, 0.5 * (np.sqrt(f ** 4 + 4 * f ** 2) - f ** 2) # RV (q=1)
177
+
178
+ def pcorr_xy(cond): # partial corr of x & target | cond
179
+ if not cond:
180
+ return float(np.corrcoef(data[x], data[target])[0, 1])
181
+ Z = np.column_stack([zsc(data[c]) for c in cond] + [np.ones(n)])
182
+ rx = zsc(data[x]) - Z @ np.linalg.lstsq(Z, zsc(data[x]), rcond=None)[0]
183
+ ry = y - Z @ np.linalg.lstsq(Z, y, rcond=None)[0]
184
+ return float(np.corrcoef(rx, ry)[0, 1])
185
+
186
+ partial, t, rv = fit(zs)
187
+ raw = float(np.corrcoef(data[x], data[target])[0, 1])
188
+ zst = ", ".join(zs) if zs else "nothing"
189
+ msg = (f"ADJUST: effect of {x} on {target} controlling for [{zst}] (n={n}). "
190
+ f"raw corr {raw:+.2f} -> adjusted partial corr {partial:+.2f} (t={t:+.1f}). "
191
+ f"Robustness value RV={rv:.2f}: an unmeasured confounder would need to explain "
192
+ f">={rv*100:.0f}% of the residual variance of BOTH {x} and {target} to null it. "
193
+ f"RV<0.10 = fragile; higher = more robust to hidden confounding.")
194
+
195
+ r0 = abs(raw); collide, explain = [], [] # per-variable bias audit
196
+ for z in zs:
197
+ rz = pcorr_xy([z])
198
+ opened = abs(rz) - r0 > 0.10 # conditioning grows the association
199
+ flipped = rz * raw < 0 and abs(rz) > 0.15 # ...or reverses its sign
200
+ if opened or flipped:
201
+ collide.append((z, rz))
202
+ elif r0 - abs(rz) > 0.15: # z soaks up much of x<->target
203
+ explain.append((z, rz))
204
+ if collide:
205
+ lst = ", ".join(f"{z} (corr {raw:+.2f}->{rz:+.2f})" for z, rz in collide)
206
+ msg += (f"\n[BIAS-AUDIT] conditioning on {lst} sharply changes the {x}-{target} link "
207
+ f"(opens or reverses it) -- a COLLIDER signature (a common effect of both). If "
208
+ f"{x} and {target} both cause it, do NOT adjust; that manufactures a spurious "
209
+ f"effect. (A strong confounder can also flip the sign -- confirm via the DAG.)")
210
+ if explain:
211
+ lst = ", ".join(f"{z} (corr {raw:+.2f}->{rz:+.2f})" for z, rz in explain)
212
+ msg += (f"\n[BIAS-AUDIT] {lst} soak(s) up much of the link. Correct to adjust ONLY if a "
213
+ f"CONFOUNDER (a prior common cause); if a MEDIATOR (on the {x}->{target} path) "
214
+ f"or measured AFTER {x}, adjusting ERASES the real effect. Data can't tell "
215
+ f"them apart -- decide by the DAG / measurement time-order.")
216
+ if zs and not collide and not explain:
217
+ msg += ("\n[BIAS-AUDIT] no collider signature in the set (still confirm none are "
218
+ "mediators / post-exposure via the DAG).")
219
+
220
+ omitted = [c for c in others if c not in zs]
221
+ if omitted: # show the full-set result as a reference
222
+ fp, ft, frv = fit(others)
223
+ flip = (abs(fp) < 0.10) != (abs(partial) < 0.10)
224
+ msg += (f"\n[FULL SET] controlling for ALL others {omitted}: {x}'s partial is {fp:+.2f} "
225
+ f"(RV={frv:.2f})" + (" -- FLIPS vs your subset." if flip else ", consistent.") +
226
+ " Trust this ONLY if none of those are mediators/colliders (see BIAS-AUDIT); for "
227
+ "observational data adjust for confounders, not 'everything'.")
228
+ return msg
229
+
230
+ def iv(x, z, controls=None):
231
+ """Instrumental-variables estimation (2SLS, hand-rolled with the same np.linalg.lstsq
232
+ primitive ADJUST already uses -- no new dependency). This is the tool for the one real
233
+ gap ADJUST cannot close: ADJUST's robustness value honestly quantifies how strong an
234
+ UNMEASURED confounder would need to be to overturn the result, but it can't rule one
235
+ out -- it only adjusts for confounders that were actually MEASURED. IV can recover the
236
+ true effect of x on target even with unmeasured confounding present, IF a valid
237
+ instrument z is available: a variable that moves x but has NO direct effect on target
238
+ except through x.
239
+ first stage: x ~ z + controls (does the instrument actually move x?)
240
+ second stage: target ~ xhat + controls (effect of x's INSTRUMENTED variation only)
241
+ Reports the first-stage F-statistic on the instrument's own coefficient (for a single
242
+ instrument this equals its partial F-test, F = t^2) -- the standard Stock-Yogo-adjacent
243
+ rule of thumb is F<10 = weak instrument, an unreliable estimate, surfaced here plainly
244
+ the same way ADJUST surfaces RV<0.10 = fragile, not buried in a docstring. A weak or
245
+ invalid instrument makes 2SLS WORSE than OLS, not better -- it does not fail safe.
246
+ The exclusion restriction (z has no direct path to target, and no unmeasured cause
247
+ shared with target) is the tool's own core assumption and CANNOT be tested from the
248
+ data handed to it -- same honest boundary as ADJUST's confounder-vs-mediator split:
249
+ data alone can't settle it, only the DAG / domain knowledge can."""
250
+ if x not in data:
251
+ return f"unknown column '{x}'; columns are {list(data)}"
252
+ if z not in data:
253
+ return f"unknown instrument '{z}'; columns are {list(data)}"
254
+ if z in (x, target):
255
+ return f"instrument must be a column other than {x} and {target} (got '{z}')."
256
+ controls = [c for c in (controls or []) if c in data and c not in (x, target, z)]
257
+ n = len(data[target])
258
+
259
+ def zsc(a):
260
+ a = np.asarray(a, float); return (a - a.mean()) / (a.std() + 1e-9)
261
+
262
+ y = zsc(data[target]); xv = zsc(data[x]); zv = zsc(data[z])
263
+ Ccols = [zsc(data[c]) for c in controls]
264
+
265
+ # first stage: x ~ z + controls + const
266
+ X1 = np.column_stack([zv] + Ccols + [np.ones(n)])
267
+ beta1, *_ = np.linalg.lstsq(X1, xv, rcond=None)
268
+ xhat = X1 @ beta1
269
+ resid1 = xv - xhat; dof1 = max(n - X1.shape[1], 1)
270
+ se1 = np.sqrt(((resid1 ** 2).sum() / dof1) * np.diag(np.linalg.pinv(X1.T @ X1)))
271
+ t_z = float(beta1[0] / (se1[0] + 1e-12))
272
+ f_stat = t_z * t_z # single excluded instrument: F == t^2
273
+
274
+ # second stage: target ~ xhat + controls + const
275
+ X2 = np.column_stack([xhat] + Ccols + [np.ones(n)])
276
+ beta2, *_ = np.linalg.lstsq(X2, y, rcond=None)
277
+ resid2 = y - X2 @ beta2; dof2 = max(n - X2.shape[1], 1)
278
+ se2 = np.sqrt(((resid2 ** 2).sum() / dof2) * np.diag(np.linalg.pinv(X2.T @ X2)))
279
+ iv_coef = float(beta2[0]); iv_t = float(beta2[0] / (se2[0] + 1e-12))
280
+
281
+ # naive OLS of target on x (+ controls), for comparison -- what ADJUST-with-no-instrument sees
282
+ Xo = np.column_stack([xv] + Ccols + [np.ones(n)])
283
+ beta_o, *_ = np.linalg.lstsq(Xo, y, rcond=None)
284
+ ols_coef = float(beta_o[0])
285
+
286
+ weak = f_stat < 10
287
+ cst = ", ".join(controls) if controls else "nothing"
288
+ msg = (f"IV (2SLS): effect of {x} on {target} using instrument {z}"
289
+ + (f", controlling for [{cst}]" if controls else "") + f" (n={n}).\n"
290
+ f" First-stage F-statistic ({z} -> {x}) = {f_stat:.1f} -- "
291
+ + ("WEAK INSTRUMENT (F<10, the standard Stock-Yogo-adjacent rule of thumb): "
292
+ "this estimate is UNRELIABLE, do not trust it." if weak else
293
+ "not weak by the standard F>=10 rule of thumb.") + "\n"
294
+ f" Naive OLS {x}->{target} (z-scored units, what ADJUST alone would see): {ols_coef:+.2f}.\n"
295
+ f" 2SLS estimate: {iv_coef:+.2f} (t={iv_t:+.1f}) -- the effect of {x} on {target} "
296
+ f"using ONLY the variation in {x} explained by {z}, robust to confounders that "
297
+ f"affect BOTH {x} and {target} but not {z}, including UNMEASURED ones.\n"
298
+ f"[EXCLUSION RESTRICTION] this assumes {z} affects {target} ONLY through {x} -- no "
299
+ f"direct effect, no unmeasured cause shared with {target}. That CANNOT be tested "
300
+ f"from this data; it must be justified by domain knowledge / the DAG, the same "
301
+ f"honest boundary as ADJUST's confounder-vs-mediator split. A weak or invalid "
302
+ f"instrument makes 2SLS WORSE than plain OLS, not better -- it does not fail safe.")
303
+ return msg
304
+
305
+ def interact(x, z):
306
+ """Tests the explicit PRODUCT term x*z as a candidate driver -- something CORR/ADJUST
307
+ structurally cannot see, by construction: a linear regression's fitted surface is
308
+ additive in its inputs, so a pure interaction (outcome driven by x*z, not x or z alone)
309
+ is invisible to it no matter how strong the real effect is. Verified directly: for
310
+ independent mean-zero x,z, corr(x,target) and corr(z,target) can both be ~0 while
311
+ corr(x*z,target) is ~1 -- the same odd/even symmetry argument as the point-group
312
+ selection rule (see github.com/tritsystem/symmetry-selection-rule): a purely
313
+ additive/linear method has no way to represent an even-order term until something
314
+ breaks that structure. Report only, not a full backdoor adjustment -- confirm x,z
315
+ aren't downstream of target before trusting this as causal, same caveat ADJUST's
316
+ bias-audit already carries."""
317
+ if x not in data or z not in data:
318
+ return f"unknown column(s); columns are {list(data)}"
319
+ rx = float(np.corrcoef(data[x], data[target])[0, 1])
320
+ rz = float(np.corrcoef(data[z], data[target])[0, 1])
321
+ product = np.asarray(data[x], float) * np.asarray(data[z], float)
322
+ rxz = float(np.corrcoef(product, data[target])[0, 1])
323
+ msg = (f"INTERACT: {x}*{z} vs {target} (n={len(data[target])}). "
324
+ f"Individually: corr({x},{target})={rx:+.2f}, corr({z},{target})={rz:+.2f}. "
325
+ f"Product term: corr({x}*{z},{target})={rxz:+.2f}.")
326
+ if abs(rxz) - max(abs(rx), abs(rz)) > 0.15:
327
+ msg += (f"\n[FOUND] the product explains far more than either variable alone -- "
328
+ f"a real candidate INTERACTION driver, invisible to CORR/ADJUST's linear-"
329
+ f"only view. Not yet a full causal claim: confirm neither {x} nor {z} is "
330
+ f"downstream of {target} before trusting this.")
331
+ else:
332
+ msg += "\nNo meaningful interaction signal beyond what the individual variables already show."
333
+ return msg
334
+
335
+ def refute(x, zs):
336
+ """DoWhy-backed refutation testing -- a SECOND, independently-derived
337
+ robustness check on the SAME x|zs hypothesis ADJUST already tested,
338
+ using an established causal-inference library's estimator plus three
339
+ real perturbation tests, instead of this file's own hand-rolled
340
+ Cinelli-Hazlett RV. Does NOT resolve confounder-vs-mediator ambiguity
341
+ (see ADJUST's bias-audit for that, still a DAG/time-order question
342
+ data alone can't answer) -- it only asks whether the NUMERICAL
343
+ estimate survives real perturbation:
344
+ placebo treatment -- effect should COLLAPSE toward 0 (confirms
345
+ the estimate isn't a fitting-procedure
346
+ artifact; treatment is randomly permuted)
347
+ random common cause -- effect should barely CHANGE (a real,
348
+ already-adjusted effect shouldn't move much
349
+ from one more irrelevant confounder)
350
+ data subset (80%) -- effect should barely CHANGE (not driven by
351
+ a handful of influential points)
352
+ Stochastic (permutation/resampling-based) -- re-running can shift the
353
+ exact numbers slightly; the qualitative collapsed/stable read is what
354
+ matters, not the third decimal place."""
355
+ if not _dowhy_available():
356
+ return "REFUTE unavailable: pip install dowhy to enable (not a core dependency)."
357
+ if x not in data:
358
+ return f"unknown column '{x}'; columns are {list(data)}"
359
+ zs = [z for z in zs if z in data and z not in (x, target)]
360
+ import pandas as pd
361
+ from dowhy import CausalModel
362
+ df = pd.DataFrame({c: data[c] for c in data})
363
+ try:
364
+ cm = CausalModel(data=df, treatment=x, outcome=target, common_causes=zs or None)
365
+ identified = cm.identify_effect(proceed_when_unidentifiable=True)
366
+ est = cm.estimate_effect(identified, method_name="backdoor.linear_regression")
367
+ orig = float(est.value)
368
+ placebo = cm.refute_estimate(identified, est, method_name="placebo_treatment_refuter", placebo_type="permute")
369
+ rcc = cm.refute_estimate(identified, est, method_name="random_common_cause")
370
+ subset = cm.refute_estimate(identified, est, method_name="data_subset_refuter", subset_fraction=0.8)
371
+ except Exception as e:
372
+ return f"REFUTE error (DoWhy could not fit/refute this model): {e}"
373
+
374
+ zst = ", ".join(zs) if zs else "nothing"
375
+ p_new, rcc_new, subset_new = float(placebo.new_effect), float(rcc.new_effect), float(subset.new_effect)
376
+ scale = abs(orig) if abs(orig) > 1e-9 else 1e-9
377
+ collapsed = abs(p_new) < 0.1 * scale
378
+ stable_rcc = abs(rcc_new - orig) < 0.2 * scale
379
+ stable_subset = abs(subset_new - orig) < 0.2 * scale
380
+ n_pass = sum([collapsed, stable_rcc, stable_subset])
381
+
382
+ msg = (f"REFUTE (DoWhy): effect of {x} on {target} controlling for [{zst}] -- "
383
+ f"original estimate {orig:+.3f} (backdoor.linear_regression, independent of ADJUST's own fit).\n"
384
+ f" placebo treatment: new effect {p_new:+.3f} -- "
385
+ + ("collapsed toward 0, as expected for a real effect." if collapsed
386
+ else "did NOT collapse -- suspicious; the estimate may reflect the fitting procedure, not a real relationship.") + "\n"
387
+ f" random common cause: new effect {rcc_new:+.3f} -- "
388
+ + ("stable." if stable_rcc else "changed notably -- sensitive to an irrelevant confounder, a fragility signal.") + "\n"
389
+ f" data subset (80%): new effect {subset_new:+.3f} -- "
390
+ + ("stable." if stable_subset else "changed notably -- may be driven by a subset of influential points.") + "\n"
391
+ f"[{n_pass}/3 refutation checks consistent with a real, stable effect]")
392
+ return msg
393
+
394
+ return corr, run, strat, adjust, interact, refute, iv
395
+
396
+ # ---------------- RECALL: optional semantic search over an external vault (OBSERVE) ----------------
397
+ _OBSERVE_PATH = os.environ.get("METHODLM_OBSERVE_PATH") # optional: path to an OBSERVE checkout
398
+ _VAULT_INDEX_DIR = os.environ.get("METHODLM_VAULT_INDEX") # optional: path to a pre-built OBSERVE index dir
399
+ _VAULT_MODEL = "sentence-transformers/all-MiniLM-L6-v2"
400
+ _VAULT_ENGINE = None # lazy singleton: loading the embedding model takes real seconds, do it once
401
+
402
+ def _vault_engine():
403
+ """RECALL is fully optional, same pattern as the tritkit second witness above: unset
404
+ either env var and RECALL just reports itself unavailable rather than erroring. False
405
+ (not None) once tried-and-failed, so a missing index doesn't retry every call."""
406
+ global _VAULT_ENGINE
407
+ if _VAULT_ENGINE is not None:
408
+ return _VAULT_ENGINE
409
+ if not _OBSERVE_PATH or not _VAULT_INDEX_DIR:
410
+ _VAULT_ENGINE = False
411
+ return _VAULT_ENGINE
412
+ if _OBSERVE_PATH not in sys.path:
413
+ sys.path.insert(0, _OBSERVE_PATH)
414
+ try:
415
+ from search_engine import SearchEngine
416
+ eng = SearchEngine()
417
+ eng.load_blocking(_VAULT_INDEX_DIR, _VAULT_MODEL)
418
+ _VAULT_ENGINE = eng if eng.ready else False
419
+ except Exception as e:
420
+ print(f"[warn] RECALL vault search unavailable: {e}")
421
+ _VAULT_ENGINE = False
422
+ return _VAULT_ENGINE
423
+
424
+ def recall(query, k=5):
425
+ """Real semantic search (MiniLM embeddings, not a keyword grep) over an external OBSERVE-
426
+ indexed corpus (e.g. a notes vault) -- 'did we already investigate/decide this before'
427
+ memory. NOT a causal test: does not count toward nrun and cannot satisfy the pre-FINAL
428
+ test requirement, same as ATTR/CORR. Fully optional -- see METHODLM_OBSERVE_PATH /
429
+ METHODLM_VAULT_INDEX above."""
430
+ eng = _vault_engine()
431
+ if not eng:
432
+ return ("RECALL unavailable: set METHODLM_OBSERVE_PATH (a checkout of OBSERVE, "
433
+ "https://github.com/tritsystem/observe-api) and METHODLM_VAULT_INDEX (a "
434
+ "pre-built semantic index directory) to enable it.")
435
+ hits = eng.search(query, k=k)
436
+ if not hits:
437
+ return f"RECALL: no vault matches for '{query}'."
438
+ lines = [f"RECALL: top {len(hits)} vault match(es) for '{query}':"]
439
+ for h in hits:
440
+ lines.append(f" [{h['score']:.2f}] {os.path.basename(h['path'])}: {h['preview'][:160].strip()}")
441
+ return "\n".join(lines)
442
+
443
+ def _split_confounders(s):
444
+ """Split an ADJUST/REFUTE/IV confounder-list argument on comma AND/OR whitespace.
445
+
446
+ Real bug, found while benchmarking the contrastive-decoding backend (a Qwen2.5-1.5B/
447
+ 0.5B pair): it reliably writes 'ADJUST: x | beta gamma delta epsilon' -- SPACE-separated,
448
+ no commas -- a plausible, common-enough LLM output style that the old `.split(",")` had
449
+ no way to handle: with no comma present the whole tail became ONE non-matching token,
450
+ got filtered out by `if z in data`, and every such ADJUST call silently ran as
451
+ 'controlling for [nothing]' (a bare, unadjusted correlation) instead of the real
452
+ adjustment -- while the tool's own [FULL SET] reference line, immune to this because it
453
+ loops confounders individually, showed the correct number a few words later in the SAME
454
+ message. It also corrupted the loop's own proactive "all candidates tested, conclude
455
+ now" nudge, which regex-greps the FIRST 'RV=' in the tool text -- the inflated raw/
456
+ unadjusted RV, not the correct adjusted one. Splitting on comma OR whitespace fixes both
457
+ silently -- no behavior change for the (still valid) comma-separated style every other
458
+ backend has been using."""
459
+ return [z for z in re.split(r"[,\s]+", s.strip()) if z]
460
+
461
+ # ---------------- REASON: the copilot loop ----------------
462
+ def build_system(cols, target, interventional):
463
+ names = ", ".join(cols)
464
+ a, b = [c for c in cols if c != target][:2]
465
+ runline = (f" RUN: vary={a}, clamp={{{b}:50}} (true controlled experiment; clamp value is a NUMBER)"
466
+ if interventional else
467
+ f" STRAT: {a},{b} (quick check: corr({a},{target}) inside bands of {b})")
468
+ return textwrap.dedent(f"""\
469
+ You are a research-methodology copilot running a real instrument by the gbranaa-hue
470
+ method. A correlation is never a cause; suspect the boring explanation (artifact,
471
+ confound, the instrument itself) first; trust only tests that could have failed.
472
+
473
+ Data columns (EXACT names): {names}. Target: {target}.
474
+ Tools -- end each reply with exactly one tool line:
475
+ CORR: {a},{target}
476
+ {runline}
477
+ ADJUST: {a} | <other confounder columns> (backdoor adjustment + sensitivity: effect
478
+ of {a} on {target} controlling for the listed confounders,
479
+ with a robustness value + a collider/mediator bias audit)
480
+ IV: {a} | instrument=<column>[, controls=<other columns>] (2SLS: use when ADJUST's
481
+ robustness value is low or a confounder is UNMEASURED --
482
+ ADJUST can only control for confounders you actually have
483
+ as columns; IV can recover the true effect even with
484
+ unmeasured confounding IF <column> is a genuine instrument
485
+ that moves {a} but has NO direct effect on {target} except
486
+ through {a}. Reports the first-stage F-statistic (F<10 =
487
+ weak instrument, unreliable estimate) and states plainly
488
+ that the exclusion restriction cannot be verified from
489
+ data alone -- same honest boundary as ADJUST's collider/
490
+ mediator split.)
491
+ INTERACT: {a},<other column> (tests the PRODUCT of two columns as a driver --
492
+ CORR/ADJUST are linear and CANNOT see a pure interaction
493
+ effect, even a perfect one, no matter how strong; if every
494
+ candidate looks fragile alone under ADJUST, try this before
495
+ concluding "no driver")
496
+ REFUTE: {a} | <same confounder columns as your ADJUST call> (SECOND, independent
497
+ robustness check on a candidate ADJUST already found
498
+ promising -- DoWhy's own estimator + 3 real perturbation
499
+ tests: placebo treatment should COLLAPSE the effect,
500
+ random common cause and data-subset should barely change
501
+ it. Use AFTER ADJUST on the same candidate, not instead of
502
+ it -- ADJUST's bias-audit and REFUTE's perturbation checks
503
+ catch different failure modes.)
504
+ ATTR: (the learned ternary model's evidence per column)
505
+ RECALL: <free-text query> (semantic search over past vault notes/decisions for
506
+ relevant prior work -- memory, NOT a causal test; use it
507
+ ONCE, if at all, to check "did we already find this" before
508
+ re-deriving -- it cannot be used a second time; unavailable
509
+ unless METHODLM_OBSERVE_PATH/METHODLM_VAULT_INDEX are set)
510
+ HOW TO FIND THE DRIVER: for a candidate X, run 'ADJUST: X | <the other candidate
511
+ columns>'. The tool shows the full-set result AND a [BIAS-AUDIT]. HEED IT: drop any
512
+ variable flagged as a COLLIDER (conditioning on it manufactures a fake effect), and do
513
+ NOT adjust for a MEDIATOR (a variable on the X->{target} path or measured after X) --
514
+ 'adjust for everything' is the Table 2 fallacy. Condition on prior common causes only.
515
+ If X's adjusted partial corr stays large with a high robustness value, X drives
516
+ {target}; if it COLLAPSES toward 0 (or RV < 0.10), X is a confounded bystander.
517
+ CRITICAL: the driver is the candidate that SURVIVES full adjustment -- it is NEVER
518
+ the variable you controlled for. Do not name a conditioning/control variable as the
519
+ cause; that is backwards. Strategy: ADJUST the tempting/decoy variable
520
+ first; if it collapses, ADJUST the other strong candidate to confirm the real
521
+ driver; once one survives, REFUTE it on the same X | <same confounders> as a second,
522
+ independently-derived check before concluding (if REFUTE is unavailable it will say so
523
+ -- don't let that block FINAL, ADJUST's own robustness value already counts as a test).
524
+ Before any ADJUST/STRAT/RUN/INTERACT/REFUTE/IV, include a 'PREREGISTER:' line in the
525
+ SAME reply naming the same X you test and what result confirms vs disconfirms. Comparing
526
+ two correlations is NOT a test. You MUST run at least one ADJUST/STRAT/RUN/INTERACT/IV
527
+ (never only CORR) before any FINAL. Your FINAL must name the variable (or product of two
528
+ variables) whose effect SURVIVED as the driver (or say none did). Be brief.""")
529
+
530
+ def ask(system, messages, n=230):
531
+ return backend().generate(system, messages, n)
532
+
533
+ def vanilla_answer(question, n=200):
534
+ """The SAME model, no method and no tools -- a plain analyst. The control racer."""
535
+ sysp = ("You are a helpful data analyst. Answer the question directly and concisely: "
536
+ "give your conclusion and a recommendation.")
537
+ return ask(sysp, [{"role": "user", "content": question}], n)
538
+
539
+ def investigate(name, data, target, question, interventional, answer_key=None, ingest_report=None):
540
+ ledger = os.path.join(HERE, f"ledger_{name}.txt")
541
+ led = open(ledger, "w", encoding="utf-8")
542
+ def out(s):
543
+ led.write(s + "\n"); led.flush()
544
+ try: print(s, flush=True)
545
+ except UnicodeEncodeError: print(s.encode("ascii", "replace").decode(), flush=True)
546
+
547
+ if ingest_report:
548
+ out(f"[ingest]\n{ingest_report}\n")
549
+ t0 = time.time()
550
+ try:
551
+ nmse, share, gate = train_readout(data, target)
552
+ top = ", ".join(f"{k} {v*100:.0f}%" for k, v in list(share.items())[:4])
553
+ out(f"[compute] ternary readout NMSE {nmse:.2f} | evidence: {top} | gate: {gate}")
554
+ except ImportError:
555
+ nmse, share, gate = None, {}, []
556
+ out("[compute] ternary second witness skipped (set METHODLM_TRITKIT + install torch to enable)")
557
+ corr, run, strat, adjust, interact, refute, iv = make_tools(data, target, interventional)
558
+ system = build_system(list(data), target, interventional)
559
+
560
+ msgs = [{"role": "user", "content": question}]
561
+ pre = nrun = n_recall = 0
562
+ verdict = "(no verdict — ran out of turns)"
563
+ seen = {} # loop guard: signatures of tests already run
564
+ stuck = 0 # consecutive turns producing no usable tool call
565
+ best_rv = {} # best (highest) RV seen per tested column -- drives the proactive
566
+ # "all candidates tested, conclude now" nudge
567
+ # 12 was the original cap. Real bug, found by rerunning a genuine 8-candidate financial
568
+ # panel (copper_xlk) after fixing the local-model context-window truncation below: the
569
+ # model correctly PREREGISTERed and ADJUSTed all 8 candidates, correctly identified the
570
+ # one survivor (spy_return, RV=0.10) and got the "[ALL CANDIDATES TESTED] ... reply FINAL
571
+ # now" nudge on turn 12 itself -- but turn 12 was the LAST turn in range(1,13), so there
572
+ # was no turn left to actually emit FINAL and the run still ended "(no verdict -- ran out
573
+ # of turns)" despite the correct answer already sitting in the transcript. A panel with
574
+ # N candidate columns needs >=N ADJUST turns plus headroom for PREREGISTER-missing/syntax
575
+ # retries (both real, observed failure modes) plus 1+ for FINAL itself -- 12 is too tight
576
+ # for anything past ~6-7 candidates. 24 gives a ~10-12 candidate panel real room without
577
+ # being unbounded (the loop-guard/no-progress-guard above still force an early stop on a
578
+ # genuinely stuck run).
579
+ for turn in range(1, 25):
580
+ raw = ask(system, msgs)
581
+ # ENFORCE ONE ACTION PER TURN: keep text up to and including the first tool
582
+ # directive (or FINAL); drop anything the model dumped after it.
583
+ cut = None
584
+ for m in re.finditer(r"^\s*(CORR|RUN|STRAT|ADJUST|INTERACT|REFUTE|IV|ATTR|RECALL|FINAL)\b.*$", raw, re.MULTILINE | re.IGNORECASE):
585
+ cut = m.end(); break
586
+ reply = raw[:cut] if cut else raw
587
+ out(f"\n--- copilot turn {turn} ---\n{reply}")
588
+ msgs.append({"role": "assistant", "content": reply})
589
+ if "PREREGISTER:" in reply: pre += 1
590
+ m_run = re.search(r"RUN:\s*vary=(\w+),\s*clamp=\{([^}]*)\}", reply)
591
+ m_adj = re.search(r"ADJUST:\s*(\w+)\s*\|\s*([\w,\s]*)", reply)
592
+ m_str = re.search(r"STRAT:\s*(\w+)\s*,\s*(\w+)", reply)
593
+ m_cor = re.search(r"CORR:\s*(\w+)\s*,\s*(\w+)", reply)
594
+ m_attr = re.search(r"\bATTR:", reply)
595
+ m_rec = re.search(r"RECALL:\s*(.+)$", reply, re.MULTILINE)
596
+ m_int = re.search(r"INTERACT:\s*(\w+)\s*,\s*(\w+)", reply)
597
+ m_ref = re.search(r"REFUTE:\s*(\w+)\s*\|\s*([\w,\s]*)", reply)
598
+ m_iv = re.search(r"IV:\s*(\w+)\s*\|\s*instrument\s*=\s*(\w+)(?:\s*,\s*controls\s*=\s*([\w,\s]*))?", reply)
599
+ # FINAL honored ONLY when it's not a hedge AND a real test has run
600
+ if re.search(r"\bfinal\s*:", reply, re.IGNORECASE) and not (m_run or m_adj or m_str or m_cor or m_attr or m_int or m_ref or m_iv):
601
+ if nrun < 1:
602
+ out("[TOOL] REFUSED: comparing correlations is not a test. Run one STRAT or "
603
+ "RUN before concluding.")
604
+ msgs.append({"role": "user", "content": "REFUSED: run at least one STRAT or "
605
+ "RUN test before FINAL."})
606
+ continue
607
+ verdict = re.sub(r"^.*?final\s*:", "", reply, flags=re.IGNORECASE | re.DOTALL).strip()
608
+ out("\n[done] verdict reached."); break
609
+ if m_run:
610
+ clamp = {k: float(v) for k, v in re.findall(r"(\w+)\s*:\s*([-\d.]+)", m_run.group(2))}
611
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
612
+ else "Clamp needs numbers, e.g. clamp={humidity:50}" if not clamp
613
+ else run(m_run.group(1), clamp)); nrun += m_run and bool(clamp) and "PREREGISTER:" in reply
614
+ elif m_adj:
615
+ zs = _split_confounders(m_adj.group(2))
616
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
617
+ else adjust(m_adj.group(1), zs)); nrun += "PREREGISTER:" in reply
618
+ elif m_str:
619
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
620
+ else strat(m_str.group(1), m_str.group(2))); nrun += bool(m_str) and "PREREGISTER:" in reply
621
+ elif m_cor:
622
+ res = corr(m_cor.group(1), m_cor.group(2))
623
+ elif m_int:
624
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
625
+ else interact(m_int.group(1), m_int.group(2))); nrun += "PREREGISTER:" in reply
626
+ elif m_ref:
627
+ zs = _split_confounders(m_ref.group(2))
628
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
629
+ else refute(m_ref.group(1), zs)); nrun += "PREREGISTER:" in reply
630
+ elif m_iv:
631
+ ctrls = _split_confounders(m_iv.group(3) or "")
632
+ res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
633
+ else iv(m_iv.group(1), m_iv.group(2), ctrls)); nrun += "PREREGISTER:" in reply
634
+ elif re.search(r"\bATTR:", reply):
635
+ if nmse is None:
636
+ res = "Ternary second witness unavailable (install torch + set METHODLM_TRITKIT)."
637
+ else:
638
+ res = (f"Ternary readout (NMSE {nmse:.2f}) evidence share: "
639
+ + ", ".join(f"{k}: {v*100:.0f}%" for k, v in share.items())
640
+ + f". Gate connects: {', '.join(gate)}.")
641
+ elif m_rec:
642
+ # SINGLE-USE GUARD: a weak model echoing the [TOOL] result text back as its own
643
+ # next RECALL: line was a real, observed failure mode (spiraling into nested
644
+ # self-quoting, e.g. "RECALL: top 5 vault match(es) for 'top 5 vault match(es)...'"
645
+ # until it lost the ability to emit any tool line at all). RECALL's job is a
646
+ # one-time "did we already do this" memory check, not a repeatable action, so cap
647
+ # it at one real call per investigation regardless of query content.
648
+ n_recall += 1
649
+ if n_recall > 1:
650
+ res = ("REFUSED: RECALL already used once this investigation -- it is a one-time "
651
+ "memory check, not repeatable. PREREGISTER and run a real causal test "
652
+ "(ADJUST/STRAT/RUN) now.")
653
+ else:
654
+ query = m_rec.group(1).strip()[:200] # cap length: reject echoed-tool-output blowup
655
+ res = recall(query)
656
+ else:
657
+ # Real observed failure mode (live UI test): the model wrote free prose containing
658
+ # "ADJUST" instead of the exact syntax ("ADJUST: effect of humidity on error
659
+ # controlling for [temperature, vibration]...") and got a generic "No tool
660
+ # recognized" that didn't teach it the fix -- it then burned the rest of its turn
661
+ # budget without ever recovering. Detecting the attempted keyword and echoing back
662
+ # the exact expected syntax gives it a real shot at self-correcting next turn.
663
+ upper = reply.upper()
664
+ if "REFUTE" in upper:
665
+ hint = " Correct REFUTE syntax: 'REFUTE: x | z1,z2' (same shape as ADJUST -- pipe before confounders)."
666
+ elif "ADJUST" in upper:
667
+ hint = " Correct ADJUST syntax: 'ADJUST: x | z1,z2' (pipe before confounders, comma-separated, no prose)."
668
+ elif "IV:" in upper:
669
+ # "IV" alone is too short to test as a bare substring (matches inside DRIVE,
670
+ # SURVIVE, POSITIVE, etc. in ordinary prose) -- require the colon the real
671
+ # syntax always has.
672
+ hint = (" Correct IV syntax: 'IV: x | instrument=z' (optionally add "
673
+ "', controls=c1,c2'); pipe before instrument=, same shape as ADJUST.")
674
+ elif "INTERACT" in upper:
675
+ hint = " Correct INTERACT syntax: 'INTERACT: x,z' (exactly two column names, comma-separated)."
676
+ elif "STRAT" in upper:
677
+ hint = " Correct STRAT syntax: 'STRAT: x,z' (exactly two column names, comma-separated)."
678
+ elif "RUN" in upper:
679
+ hint = " Correct RUN syntax: 'RUN: vary=x, clamp={z:NUMBER}' (clamp value must be a number)."
680
+ elif "CORR" in upper:
681
+ hint = " Correct CORR syntax: 'CORR: x,y' (exactly two column names, comma-separated)."
682
+ else:
683
+ hint = ""
684
+ res = f"No tool recognized.{hint} Use CORR:, STRAT:/RUN:, ADJUST:, IV:, INTERACT:, REFUTE:, ATTR:, RECALL:, or FINAL:."
685
+ # LOOP GUARD: a weak model can re-run the same test forever. If a real test
686
+ # repeats, don't re-run it -- nudge to conclude; force a stop on a 2nd repeat.
687
+ # Tagged with the tool name: REFUTE is DESIGNED to be re-run on the exact same
688
+ # (x, zs) an ADJUST call already used (a second, independent robustness check
689
+ # on the same candidate) -- an untagged sig would wrongly flag that as a repeat.
690
+ _sig_src = [("RUN", m_run), ("ADJUST", m_adj), ("STRAT", m_str), ("CORR", m_cor),
691
+ ("INTERACT", m_int), ("REFUTE", m_ref), ("IV", m_iv)]
692
+ sig = next(((tag, g.groups()) for tag, g in _sig_src if g), None)
693
+ if sig and not str(res).startswith(("REFUSED", "Clamp")):
694
+ # Track the best (highest) RV seen per tested column -- real bug, found by
695
+ # running this exact scenario: the model found the correct driver (humidity,
696
+ # RV=0.42) at turn 6, ruled out the only other candidate at turn 7, then instead
697
+ # of concluding, re-ran an already-collapsed test a 3rd time and got hard-stopped
698
+ # by the loop-guard with NO verdict, despite the right answer already sitting in
699
+ # the transcript. The "all candidates tested, none left" signal already existed
700
+ # but only fired reactively inside the repeat-branch below -- one turn too late
701
+ # to have prevented the wasted turn-8 retry. Checking it proactively, right after
702
+ # the test that completes coverage, gives the model the strongest signal a full
703
+ # turn earlier. sig[1][0] (not sig[0]) -- sig is (tag, groups) here, tagged so
704
+ # REFUTE re-testing the same candidate ADJUST already used isn't misread as a repeat.
705
+ rv_match = re.search(r"RV=([\d.]+)", str(res))
706
+ if rv_match:
707
+ col = sig[1][0]
708
+ best_rv[col] = max(best_rv.get(col, 0.0), float(rv_match.group(1)))
709
+ all_candidates = [c for c in data if c != target]
710
+ if set(best_rv) >= set(all_candidates):
711
+ survivors = [c for c, rv in best_rv.items() if rv >= 0.10]
712
+ bystanders = [c for c in all_candidates if c not in survivors]
713
+ if survivors:
714
+ res = (f"{res}\n[ALL CANDIDATES TESTED] Survivor(s) with RV>=0.10: "
715
+ f"{', '.join(survivors)}. Bystander(s) with RV<0.10: "
716
+ f"{', '.join(bystanders) or 'none'}. You have enough evidence -- "
717
+ f"reply FINAL now naming the survivor as the driver. Do not run "
718
+ f"another test.")
719
+ if sig in seen:
720
+ seen[sig] += 1
721
+ # nudge toward the NEXT untested candidate (not FINAL) -- a collapsed test
722
+ # means that variable is a bystander, not that the job is done.
723
+ tested = ", ".join(sorted({s[1][0] for s in seen})) or "none"
724
+ untested = [c for c in data if c not in (target,) and c not in {s[1][0] for s in seen}]
725
+ res = (f"{res}\n[REPEAT: you already ran this. A collapsed effect (RV<0.10) means "
726
+ f"that variable is a BYSTANDER, not the answer. You have tested: {tested}. "
727
+ f"Now run ADJUST on a DIFFERENT untested candidate ({', '.join(untested) or 'none left'}) "
728
+ "to find the real driver. Reply FINAL only once a variable SURVIVES (high RV).]")
729
+ if seen[sig] >= 3:
730
+ out(f"[TOOL] {res}")
731
+ verdict = "(loop-guard: model repeated the same test without concluding)"
732
+ out("\n[done] loop-guard stopped a repeat loop."); break
733
+ else:
734
+ seen[sig] = 1
735
+ out(f"[TOOL] {res}")
736
+ # NO-PROGRESS guard: a model that can't emit tool syntax loops uselessly.
737
+ if str(res).startswith(("No tool recognized", "REFUSED")):
738
+ stuck += 1
739
+ if stuck >= 3:
740
+ verdict = "(model could not drive the tool protocol)"
741
+ out("\n[done] no-progress guard: model never produced a usable test."); break
742
+ else:
743
+ stuck = 0
744
+ msgs.append({"role": "user", "content": res})
745
+ if answer_key:
746
+ out(f"\n[ANSWER KEY -- the instrument was never told] {answer_key}")
747
+ out(f"\n[ledger] preregistrations {pre}, registered tests {nrun}, "
748
+ f"{time.time()-t0:.0f}s -> {ledger}")
749
+ led.close()
750
+ return {"verdict": verdict, "pre": pre, "nrun": nrun,
751
+ "nmse": nmse, "gate": gate, "share": share}
752
+
753
+
754
+ def finish_race(question, res, answer_key=None):
755
+ """Same question, two racers: MethodLM (method + tools + tests) vs plain LLM."""
756
+ print("\n" + "=" * 62)
757
+ print("HEAD-TO-HEAD -- same question, two racers")
758
+ print("=" * 62)
759
+ v = vanilla_answer(question)
760
+ tested = f"{res['nrun']} test(s), {res['pre']} pre-reg" if res["nrun"] else "NO TEST RUN"
761
+ print(f"\n[ vanilla 3B | no method, no tools ]\n {v}\n")
762
+ print(f"[ MethodLM | method + tools | {tested} ]\n {res['verdict']}\n")
763
+ if answer_key:
764
+ print(f"[ answer key ] {answer_key}")
765
+
766
+ def main():
767
+ ap = argparse.ArgumentParser()
768
+ ap.add_argument("--demo", action="store_true")
769
+ ap.add_argument("--diabetes", action="store_true")
770
+ ap.add_argument("--data", help="any file: csv/tsv/json/jsonl/parquet/sqlite/npz/xlsx")
771
+ ap.add_argument("--csv", help="alias for --data")
772
+ ap.add_argument("--folder", help="labelled folder of text/image files")
773
+ ap.add_argument("--target")
774
+ ap.add_argument("--table", help="sqlite table name (optional)")
775
+ ap.add_argument("--query", help="sqlite SQL query (optional)")
776
+ ap.add_argument("--race", action="store_true", help="also run a plain-LLM racer on the same question")
777
+ ap.add_argument("--model", default="local", help="reasoning backend: local (Qwen-3B) | claude")
778
+ args = ap.parse_args()
779
+ global BACKEND
780
+ BACKEND = methodlm_models.get_model(args.model, HERE)
781
+ print(f"[methodlm] reasoning backend: {BACKEND.label}")
782
+ if args.demo:
783
+ data = demo_world()
784
+ r = float(np.corrcoef(data['temperature'], data['error'])[0, 1])
785
+ q = (f"Our sensor's error correlates with temperature (r={r:+.2f}) in 2000 logged "
786
+ "samples; the team wants to install cooling. Find what actually drives the "
787
+ "error before we spend the money.")
788
+ key = "error = f(humidity); temperature only co-rises with humidity via season."
789
+ res = investigate("demo", data, "error", q, True, key)
790
+ if args.race: finish_race(q, res, key)
791
+ elif args.diabetes:
792
+ data = load_diabetes()
793
+ r = float(np.corrcoef(data['bmi'], data['progression'])[0, 1])
794
+ q = (f"In 442 real diabetes patients, bmi correlates with one-year disease "
795
+ f"progression (r={r:+.2f}). The clinic wants to fund a weight-loss-only "
796
+ "program. Before they do: is bmi's link robust, or does it run through blood "
797
+ "serum markers like ltg? You cannot rerun patients; use STRAT.")
798
+ res = investigate("diabetes", data, "progression", q, False)
799
+ if args.race: finish_race(q, res)
800
+ elif args.folder and args.target:
801
+ from methodlm_io import load_folder, featurize, format_report
802
+ raw, notes = load_folder(args.folder)
803
+ data, rep = featurize(raw, args.target)
804
+ report = format_report(notes, rep, args.target)
805
+ cols = [c for c in data if c != args.target]
806
+ q = (f"Investigate what actually drives {args.target} across these files "
807
+ f"(features: {', '.join(cols[:12])}). Do not trust raw correlations.")
808
+ res = investigate(os.path.basename(args.folder.rstrip('/\\')) or "folder",
809
+ data, args.target, q, False, ingest_report=report)
810
+ if args.race: finish_race(q, res)
811
+ elif args.data or args.csv:
812
+ from methodlm_io import validate, load_any, featurize, format_report
813
+ path = args.data or args.csv
814
+ v = validate(path, args.target, table=args.table, query=args.query)
815
+ if not v["ok"]:
816
+ print("MethodLM cannot run yet:")
817
+ for e in v["errors"]:
818
+ print(f" x {e}")
819
+ if v["suggestions"]:
820
+ print(f" -> try: {', '.join(map(str, v['suggestions']))}")
821
+ if v["info"]:
822
+ print(" columns available: "
823
+ + ", ".join(f"{c['name']} [{c['kind']}]" for c in v["info"]["columns"]))
824
+ return
825
+ raw, notes = load_any(path, table=args.table, query=args.query)
826
+ data, rep = featurize(raw, args.target)
827
+ report = format_report(notes, rep, args.target)
828
+ cols = [c for c in data if c != args.target]
829
+ q = (f"Investigate what actually drives {args.target} in this recorded dataset "
830
+ f"(columns: {', '.join(cols[:12])}{'...' if len(cols) > 12 else ''}). "
831
+ "Do not trust raw correlations.")
832
+ name = os.path.splitext(os.path.basename(path))[0]
833
+ res = investigate(name, data, args.target, q, False, ingest_report=report)
834
+ if args.race: finish_race(q, res)
835
+ else:
836
+ ap.print_help()
837
+
838
+ if __name__ == "__main__":
839
+ main()