methodlm 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- benchmark_causal.py +119 -0
- benchmark_models.py +224 -0
- benchmark_real_examples.py +452 -0
- methodlm-1.0.1.dist-info/METADATA +155 -0
- methodlm-1.0.1.dist-info/RECORD +14 -0
- methodlm-1.0.1.dist-info/WHEEL +5 -0
- methodlm-1.0.1.dist-info/entry_points.txt +3 -0
- methodlm-1.0.1.dist-info/licenses/LICENSE +21 -0
- methodlm-1.0.1.dist-info/top_level.txt +8 -0
- methodlm.py +839 -0
- methodlm_gui.py +103 -0
- methodlm_io.py +246 -0
- methodlm_models.py +294 -0
- rescore.py +17 -0
methodlm.py
ADDED
|
@@ -0,0 +1,839 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""MethodLM -- the method, in a language model, kept honest by a ledger.
|
|
3
|
+
|
|
4
|
+
Point it at data; it computes on it (ternary two-timescale readout) and reasons about
|
|
5
|
+
it (gbranaa-hue method), pre-registering every test.
|
|
6
|
+
|
|
7
|
+
python methodlm.py --demo interventional demo world (hidden
|
|
8
|
+
confound + answer key at the end)
|
|
9
|
+
python methodlm.py --diabetes real clinical data (442 patients,
|
|
10
|
+
sklearn load_diabetes, raw units)
|
|
11
|
+
python methodlm.py --csv F --target COL any recorded CSV (observational)
|
|
12
|
+
|
|
13
|
+
Pipeline (identical in all modes):
|
|
14
|
+
COMPUTE a tritkit TwoTimescaleLinear ternary readout learns target from the other
|
|
15
|
+
columns -> NMSE + structural evidence share + gate selection.
|
|
16
|
+
REASON the method copilot (Qwen-3B + gbranaa-hue method) investigates with tools,
|
|
17
|
+
pre-registers every test, writes an honest ledger.
|
|
18
|
+
|
|
19
|
+
Tools by mode: CORR (all) | ATTR (all) | RUN true intervention (demo world only)
|
|
20
|
+
STRAT observational conditioning (recorded data): corr(x,target)
|
|
21
|
+
inside quartile bands of z -- the honest substitute for clamping
|
|
22
|
+
when you cannot rerun the world.
|
|
23
|
+
"""
|
|
24
|
+
import argparse, os, re, sys, subprocess, textwrap, time
|
|
25
|
+
import numpy as np
|
|
26
|
+
|
|
27
|
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
28
|
+
_tk = os.environ.get("METHODLM_TRITKIT") # optional: path to the tritkit pkg (ternary 2nd witness)
|
|
29
|
+
if _tk:
|
|
30
|
+
sys.path.insert(0, _tk)
|
|
31
|
+
import methodlm_models
|
|
32
|
+
rng = np.random.default_rng(11)
|
|
33
|
+
|
|
34
|
+
BACKEND = None # the model driving the reasoning; set by main() or lazily to local
|
|
35
|
+
def backend():
|
|
36
|
+
global BACKEND
|
|
37
|
+
if BACKEND is None:
|
|
38
|
+
BACKEND = methodlm_models.get_model(os.environ.get("METHODLM_MODEL", "local"), HERE)
|
|
39
|
+
return BACKEND
|
|
40
|
+
|
|
41
|
+
# ---------------- data sources ----------------
|
|
42
|
+
def demo_world(n=2000, clamp=None):
|
|
43
|
+
clamp = clamp or {}
|
|
44
|
+
season = np.cumsum(rng.normal(0, 0.15, n)); season -= season.mean()
|
|
45
|
+
d = {"temperature": 25 + 6 * np.tanh(season) + rng.normal(0, 0.8, n),
|
|
46
|
+
"humidity": 50 + 15 * np.tanh(season) + rng.normal(0, 2.5, n),
|
|
47
|
+
"vibration": rng.normal(0, 1, n)}
|
|
48
|
+
for k, v in clamp.items():
|
|
49
|
+
if k in d: d[k] = np.full(n, float(v))
|
|
50
|
+
d["error"] = np.clip(0.5 + 0.08 * (d["humidity"] - 50) + rng.normal(0, 0.35, n), 0, None)
|
|
51
|
+
return d
|
|
52
|
+
|
|
53
|
+
def load_diabetes():
|
|
54
|
+
from sklearn.datasets import load_diabetes as ld
|
|
55
|
+
raw = ld(scaled=False)
|
|
56
|
+
names = ["age", "sex", "bmi", "bp", "tc", "ldl", "hdl", "tch", "ltg", "glu"]
|
|
57
|
+
d = {n: raw.data[:, i].astype(float) for i, n in enumerate(names)}
|
|
58
|
+
d["progression"] = raw.target.astype(float)
|
|
59
|
+
return d
|
|
60
|
+
|
|
61
|
+
def load_csv(path, target):
|
|
62
|
+
import csv as _csv
|
|
63
|
+
with open(path, newline="", encoding="utf-8-sig") as f:
|
|
64
|
+
rows = list(_csv.DictReader(f))
|
|
65
|
+
cols = {}
|
|
66
|
+
for k in rows[0]:
|
|
67
|
+
try:
|
|
68
|
+
cols[k.strip()] = np.array([float(r[k]) for r in rows])
|
|
69
|
+
except ValueError:
|
|
70
|
+
pass # skip non-numeric columns
|
|
71
|
+
assert target in cols, f"target '{target}' not among numeric columns {list(cols)}"
|
|
72
|
+
return cols
|
|
73
|
+
|
|
74
|
+
# ---------------- COMPUTE: ternary two-timescale readout ----------------
|
|
75
|
+
def train_readout(data, target):
|
|
76
|
+
import torch
|
|
77
|
+
import torch.nn.functional as F
|
|
78
|
+
from tritkit.twotimescale import TwoTimescaleLinear
|
|
79
|
+
torch.manual_seed(0)
|
|
80
|
+
names = [k for k in data if k != target]
|
|
81
|
+
X = np.column_stack([data[k] for k in names])
|
|
82
|
+
X = (X - X.mean(0)) / (X.std(0) + 1e-9)
|
|
83
|
+
Yv = data[target]; Yn = (Yv - Yv.mean()) / Yv.std()
|
|
84
|
+
Xt, Yt = torch.tensor(X, dtype=torch.float32), torch.tensor(Yn, dtype=torch.float32)
|
|
85
|
+
n = len(Yn)
|
|
86
|
+
lyr = TwoTimescaleLinear(len(names), 1, density=min(0.5, 3 / len(names) + 0.15),
|
|
87
|
+
bias=True, evidence_beta=0.97)
|
|
88
|
+
opt = torch.optim.SGD(lyr.parameters(), lr=0.02)
|
|
89
|
+
e2 = []
|
|
90
|
+
for ep in range(10):
|
|
91
|
+
perm = torch.randperm(n)
|
|
92
|
+
for t in range(0, n - 16, 16):
|
|
93
|
+
idx = perm[t:t + 16]
|
|
94
|
+
loss = F.mse_loss(lyr(Xt[idx]), Yt[idx, None])
|
|
95
|
+
opt.zero_grad(); loss.backward(); opt.step()
|
|
96
|
+
if t % 320 == 0: lyr.step_gate()
|
|
97
|
+
if ep == 9: e2.append(loss.item())
|
|
98
|
+
ev = lyr.evidence[0].detach().numpy(); share = ev / ev.sum()
|
|
99
|
+
order = np.argsort(-share)
|
|
100
|
+
return (float(np.mean(e2)),
|
|
101
|
+
{names[i]: float(share[i]) for i in order},
|
|
102
|
+
[names[i] for i in range(len(names)) if lyr.G[0, i] > 0])
|
|
103
|
+
|
|
104
|
+
# ---------------- tools ----------------
|
|
105
|
+
_DOWHY_OK = None
|
|
106
|
+
def _dowhy_available():
|
|
107
|
+
"""Same lazy-singleton optional-dependency pattern as RECALL's
|
|
108
|
+
_vault_engine() below: tried-and-failed is cached (True/False), not
|
|
109
|
+
retried every call."""
|
|
110
|
+
global _DOWHY_OK
|
|
111
|
+
if _DOWHY_OK is None:
|
|
112
|
+
try:
|
|
113
|
+
import dowhy # noqa: F401
|
|
114
|
+
_DOWHY_OK = True
|
|
115
|
+
except ImportError:
|
|
116
|
+
_DOWHY_OK = False
|
|
117
|
+
return _DOWHY_OK
|
|
118
|
+
|
|
119
|
+
def make_tools(data, target, interventional):
|
|
120
|
+
def corr(a, b):
|
|
121
|
+
if a not in data or b not in data:
|
|
122
|
+
return f"unknown column(s); columns are {list(data)}"
|
|
123
|
+
r = float(np.corrcoef(data[a], data[b])[0, 1])
|
|
124
|
+
return f"corr({a},{b}) = {r:+.2f} over {len(data[a])} samples."
|
|
125
|
+
|
|
126
|
+
def run(vary, clamp):
|
|
127
|
+
if not interventional:
|
|
128
|
+
return "RUN unavailable: this is recorded data, not a rerunnable system. Use STRAT."
|
|
129
|
+
d = demo_world(400, clamp=clamp)
|
|
130
|
+
r = float(np.corrcoef(d[vary], d[target])[0, 1]) if vary in d else float("nan")
|
|
131
|
+
cl = ", ".join(f"{k}={v}" for k, v in clamp.items()) or "nothing"
|
|
132
|
+
return (f"Controlled run: 400 fresh trials varying {vary}, clamping {cl}. "
|
|
133
|
+
f"corr({vary},{target}) = {r:+.2f}; mean {target} {d[target].mean():.2f}.")
|
|
134
|
+
|
|
135
|
+
def strat(x, z):
|
|
136
|
+
if x not in data or z not in data:
|
|
137
|
+
return f"unknown column(s); columns are {list(data)}"
|
|
138
|
+
raw = float(np.corrcoef(data[x], data[target])[0, 1])
|
|
139
|
+
qs = np.quantile(data[z], [0, .25, .5, .75, 1.0])
|
|
140
|
+
rs = []
|
|
141
|
+
for i in range(4):
|
|
142
|
+
m = (data[z] >= qs[i]) & (data[z] <= qs[i + 1])
|
|
143
|
+
if m.sum() > 20 and np.std(data[x][m]) > 0:
|
|
144
|
+
rs.append(float(np.corrcoef(data[x][m], data[target][m])[0, 1]))
|
|
145
|
+
within = float(np.mean(rs)) if rs else float("nan")
|
|
146
|
+
return (f"Stratified: raw corr({x},{target}) = {raw:+.2f}; inside quartile bands of {z} "
|
|
147
|
+
f"it is {['%+.2f' % r for r in rs]} (mean {within:+.2f}). "
|
|
148
|
+
f"If the within-band mean collapses, {x}'s link runs through {z}.")
|
|
149
|
+
|
|
150
|
+
def adjust(x, zs):
|
|
151
|
+
"""Backdoor adjustment (multiple regression) + Cinelli-Hazlett sensitivity, WITH a
|
|
152
|
+
collider/mediator bias audit. Adjusting for a MEDIATOR (on the X->target path) or a
|
|
153
|
+
COLLIDER (a common effect of X and target) INTRODUCES bias -- the 'Table 2 fallacy' --
|
|
154
|
+
so 'control for everything' is wrong for observational/causal data. Data alone cannot
|
|
155
|
+
prove a variable's role (a confounder and a mediator are observationally identical);
|
|
156
|
+
this flags the one danger that IS detectable (a collider: conditioning opens a path
|
|
157
|
+
and RAISES the X-target association) and defers the rest to the DAG / time-order."""
|
|
158
|
+
if x not in data:
|
|
159
|
+
return f"unknown column '{x}'; columns are {list(data)}"
|
|
160
|
+
others = [c for c in data if c not in (x, target)]
|
|
161
|
+
zs = [z for z in zs if z in data and z not in (x, target)]
|
|
162
|
+
n = len(data[target])
|
|
163
|
+
|
|
164
|
+
def zsc(a):
|
|
165
|
+
a = np.asarray(a, float); return (a - a.mean()) / (a.std() + 1e-9)
|
|
166
|
+
y = zsc(data[target])
|
|
167
|
+
|
|
168
|
+
def fit(cond): # partial corr, t, RV of x | cond
|
|
169
|
+
X = np.column_stack([zsc(data[c]) for c in [x] + cond] + [np.ones(n)])
|
|
170
|
+
beta, *_ = np.linalg.lstsq(X, y, rcond=None)
|
|
171
|
+
resid = y - X @ beta; dof = n - X.shape[1]
|
|
172
|
+
se = np.sqrt(((resid ** 2).sum() / max(dof, 1)) * np.diag(np.linalg.pinv(X.T @ X)))
|
|
173
|
+
t = float(beta[0] / (se[0] + 1e-12))
|
|
174
|
+
partial = t / np.sqrt(t * t + dof) if dof > 0 else float("nan")
|
|
175
|
+
f = abs(t) / np.sqrt(max(dof, 1))
|
|
176
|
+
return partial, t, 0.5 * (np.sqrt(f ** 4 + 4 * f ** 2) - f ** 2) # RV (q=1)
|
|
177
|
+
|
|
178
|
+
def pcorr_xy(cond): # partial corr of x & target | cond
|
|
179
|
+
if not cond:
|
|
180
|
+
return float(np.corrcoef(data[x], data[target])[0, 1])
|
|
181
|
+
Z = np.column_stack([zsc(data[c]) for c in cond] + [np.ones(n)])
|
|
182
|
+
rx = zsc(data[x]) - Z @ np.linalg.lstsq(Z, zsc(data[x]), rcond=None)[0]
|
|
183
|
+
ry = y - Z @ np.linalg.lstsq(Z, y, rcond=None)[0]
|
|
184
|
+
return float(np.corrcoef(rx, ry)[0, 1])
|
|
185
|
+
|
|
186
|
+
partial, t, rv = fit(zs)
|
|
187
|
+
raw = float(np.corrcoef(data[x], data[target])[0, 1])
|
|
188
|
+
zst = ", ".join(zs) if zs else "nothing"
|
|
189
|
+
msg = (f"ADJUST: effect of {x} on {target} controlling for [{zst}] (n={n}). "
|
|
190
|
+
f"raw corr {raw:+.2f} -> adjusted partial corr {partial:+.2f} (t={t:+.1f}). "
|
|
191
|
+
f"Robustness value RV={rv:.2f}: an unmeasured confounder would need to explain "
|
|
192
|
+
f">={rv*100:.0f}% of the residual variance of BOTH {x} and {target} to null it. "
|
|
193
|
+
f"RV<0.10 = fragile; higher = more robust to hidden confounding.")
|
|
194
|
+
|
|
195
|
+
r0 = abs(raw); collide, explain = [], [] # per-variable bias audit
|
|
196
|
+
for z in zs:
|
|
197
|
+
rz = pcorr_xy([z])
|
|
198
|
+
opened = abs(rz) - r0 > 0.10 # conditioning grows the association
|
|
199
|
+
flipped = rz * raw < 0 and abs(rz) > 0.15 # ...or reverses its sign
|
|
200
|
+
if opened or flipped:
|
|
201
|
+
collide.append((z, rz))
|
|
202
|
+
elif r0 - abs(rz) > 0.15: # z soaks up much of x<->target
|
|
203
|
+
explain.append((z, rz))
|
|
204
|
+
if collide:
|
|
205
|
+
lst = ", ".join(f"{z} (corr {raw:+.2f}->{rz:+.2f})" for z, rz in collide)
|
|
206
|
+
msg += (f"\n[BIAS-AUDIT] conditioning on {lst} sharply changes the {x}-{target} link "
|
|
207
|
+
f"(opens or reverses it) -- a COLLIDER signature (a common effect of both). If "
|
|
208
|
+
f"{x} and {target} both cause it, do NOT adjust; that manufactures a spurious "
|
|
209
|
+
f"effect. (A strong confounder can also flip the sign -- confirm via the DAG.)")
|
|
210
|
+
if explain:
|
|
211
|
+
lst = ", ".join(f"{z} (corr {raw:+.2f}->{rz:+.2f})" for z, rz in explain)
|
|
212
|
+
msg += (f"\n[BIAS-AUDIT] {lst} soak(s) up much of the link. Correct to adjust ONLY if a "
|
|
213
|
+
f"CONFOUNDER (a prior common cause); if a MEDIATOR (on the {x}->{target} path) "
|
|
214
|
+
f"or measured AFTER {x}, adjusting ERASES the real effect. Data can't tell "
|
|
215
|
+
f"them apart -- decide by the DAG / measurement time-order.")
|
|
216
|
+
if zs and not collide and not explain:
|
|
217
|
+
msg += ("\n[BIAS-AUDIT] no collider signature in the set (still confirm none are "
|
|
218
|
+
"mediators / post-exposure via the DAG).")
|
|
219
|
+
|
|
220
|
+
omitted = [c for c in others if c not in zs]
|
|
221
|
+
if omitted: # show the full-set result as a reference
|
|
222
|
+
fp, ft, frv = fit(others)
|
|
223
|
+
flip = (abs(fp) < 0.10) != (abs(partial) < 0.10)
|
|
224
|
+
msg += (f"\n[FULL SET] controlling for ALL others {omitted}: {x}'s partial is {fp:+.2f} "
|
|
225
|
+
f"(RV={frv:.2f})" + (" -- FLIPS vs your subset." if flip else ", consistent.") +
|
|
226
|
+
" Trust this ONLY if none of those are mediators/colliders (see BIAS-AUDIT); for "
|
|
227
|
+
"observational data adjust for confounders, not 'everything'.")
|
|
228
|
+
return msg
|
|
229
|
+
|
|
230
|
+
def iv(x, z, controls=None):
|
|
231
|
+
"""Instrumental-variables estimation (2SLS, hand-rolled with the same np.linalg.lstsq
|
|
232
|
+
primitive ADJUST already uses -- no new dependency). This is the tool for the one real
|
|
233
|
+
gap ADJUST cannot close: ADJUST's robustness value honestly quantifies how strong an
|
|
234
|
+
UNMEASURED confounder would need to be to overturn the result, but it can't rule one
|
|
235
|
+
out -- it only adjusts for confounders that were actually MEASURED. IV can recover the
|
|
236
|
+
true effect of x on target even with unmeasured confounding present, IF a valid
|
|
237
|
+
instrument z is available: a variable that moves x but has NO direct effect on target
|
|
238
|
+
except through x.
|
|
239
|
+
first stage: x ~ z + controls (does the instrument actually move x?)
|
|
240
|
+
second stage: target ~ xhat + controls (effect of x's INSTRUMENTED variation only)
|
|
241
|
+
Reports the first-stage F-statistic on the instrument's own coefficient (for a single
|
|
242
|
+
instrument this equals its partial F-test, F = t^2) -- the standard Stock-Yogo-adjacent
|
|
243
|
+
rule of thumb is F<10 = weak instrument, an unreliable estimate, surfaced here plainly
|
|
244
|
+
the same way ADJUST surfaces RV<0.10 = fragile, not buried in a docstring. A weak or
|
|
245
|
+
invalid instrument makes 2SLS WORSE than OLS, not better -- it does not fail safe.
|
|
246
|
+
The exclusion restriction (z has no direct path to target, and no unmeasured cause
|
|
247
|
+
shared with target) is the tool's own core assumption and CANNOT be tested from the
|
|
248
|
+
data handed to it -- same honest boundary as ADJUST's confounder-vs-mediator split:
|
|
249
|
+
data alone can't settle it, only the DAG / domain knowledge can."""
|
|
250
|
+
if x not in data:
|
|
251
|
+
return f"unknown column '{x}'; columns are {list(data)}"
|
|
252
|
+
if z not in data:
|
|
253
|
+
return f"unknown instrument '{z}'; columns are {list(data)}"
|
|
254
|
+
if z in (x, target):
|
|
255
|
+
return f"instrument must be a column other than {x} and {target} (got '{z}')."
|
|
256
|
+
controls = [c for c in (controls or []) if c in data and c not in (x, target, z)]
|
|
257
|
+
n = len(data[target])
|
|
258
|
+
|
|
259
|
+
def zsc(a):
|
|
260
|
+
a = np.asarray(a, float); return (a - a.mean()) / (a.std() + 1e-9)
|
|
261
|
+
|
|
262
|
+
y = zsc(data[target]); xv = zsc(data[x]); zv = zsc(data[z])
|
|
263
|
+
Ccols = [zsc(data[c]) for c in controls]
|
|
264
|
+
|
|
265
|
+
# first stage: x ~ z + controls + const
|
|
266
|
+
X1 = np.column_stack([zv] + Ccols + [np.ones(n)])
|
|
267
|
+
beta1, *_ = np.linalg.lstsq(X1, xv, rcond=None)
|
|
268
|
+
xhat = X1 @ beta1
|
|
269
|
+
resid1 = xv - xhat; dof1 = max(n - X1.shape[1], 1)
|
|
270
|
+
se1 = np.sqrt(((resid1 ** 2).sum() / dof1) * np.diag(np.linalg.pinv(X1.T @ X1)))
|
|
271
|
+
t_z = float(beta1[0] / (se1[0] + 1e-12))
|
|
272
|
+
f_stat = t_z * t_z # single excluded instrument: F == t^2
|
|
273
|
+
|
|
274
|
+
# second stage: target ~ xhat + controls + const
|
|
275
|
+
X2 = np.column_stack([xhat] + Ccols + [np.ones(n)])
|
|
276
|
+
beta2, *_ = np.linalg.lstsq(X2, y, rcond=None)
|
|
277
|
+
resid2 = y - X2 @ beta2; dof2 = max(n - X2.shape[1], 1)
|
|
278
|
+
se2 = np.sqrt(((resid2 ** 2).sum() / dof2) * np.diag(np.linalg.pinv(X2.T @ X2)))
|
|
279
|
+
iv_coef = float(beta2[0]); iv_t = float(beta2[0] / (se2[0] + 1e-12))
|
|
280
|
+
|
|
281
|
+
# naive OLS of target on x (+ controls), for comparison -- what ADJUST-with-no-instrument sees
|
|
282
|
+
Xo = np.column_stack([xv] + Ccols + [np.ones(n)])
|
|
283
|
+
beta_o, *_ = np.linalg.lstsq(Xo, y, rcond=None)
|
|
284
|
+
ols_coef = float(beta_o[0])
|
|
285
|
+
|
|
286
|
+
weak = f_stat < 10
|
|
287
|
+
cst = ", ".join(controls) if controls else "nothing"
|
|
288
|
+
msg = (f"IV (2SLS): effect of {x} on {target} using instrument {z}"
|
|
289
|
+
+ (f", controlling for [{cst}]" if controls else "") + f" (n={n}).\n"
|
|
290
|
+
f" First-stage F-statistic ({z} -> {x}) = {f_stat:.1f} -- "
|
|
291
|
+
+ ("WEAK INSTRUMENT (F<10, the standard Stock-Yogo-adjacent rule of thumb): "
|
|
292
|
+
"this estimate is UNRELIABLE, do not trust it." if weak else
|
|
293
|
+
"not weak by the standard F>=10 rule of thumb.") + "\n"
|
|
294
|
+
f" Naive OLS {x}->{target} (z-scored units, what ADJUST alone would see): {ols_coef:+.2f}.\n"
|
|
295
|
+
f" 2SLS estimate: {iv_coef:+.2f} (t={iv_t:+.1f}) -- the effect of {x} on {target} "
|
|
296
|
+
f"using ONLY the variation in {x} explained by {z}, robust to confounders that "
|
|
297
|
+
f"affect BOTH {x} and {target} but not {z}, including UNMEASURED ones.\n"
|
|
298
|
+
f"[EXCLUSION RESTRICTION] this assumes {z} affects {target} ONLY through {x} -- no "
|
|
299
|
+
f"direct effect, no unmeasured cause shared with {target}. That CANNOT be tested "
|
|
300
|
+
f"from this data; it must be justified by domain knowledge / the DAG, the same "
|
|
301
|
+
f"honest boundary as ADJUST's confounder-vs-mediator split. A weak or invalid "
|
|
302
|
+
f"instrument makes 2SLS WORSE than plain OLS, not better -- it does not fail safe.")
|
|
303
|
+
return msg
|
|
304
|
+
|
|
305
|
+
def interact(x, z):
|
|
306
|
+
"""Tests the explicit PRODUCT term x*z as a candidate driver -- something CORR/ADJUST
|
|
307
|
+
structurally cannot see, by construction: a linear regression's fitted surface is
|
|
308
|
+
additive in its inputs, so a pure interaction (outcome driven by x*z, not x or z alone)
|
|
309
|
+
is invisible to it no matter how strong the real effect is. Verified directly: for
|
|
310
|
+
independent mean-zero x,z, corr(x,target) and corr(z,target) can both be ~0 while
|
|
311
|
+
corr(x*z,target) is ~1 -- the same odd/even symmetry argument as the point-group
|
|
312
|
+
selection rule (see github.com/tritsystem/symmetry-selection-rule): a purely
|
|
313
|
+
additive/linear method has no way to represent an even-order term until something
|
|
314
|
+
breaks that structure. Report only, not a full backdoor adjustment -- confirm x,z
|
|
315
|
+
aren't downstream of target before trusting this as causal, same caveat ADJUST's
|
|
316
|
+
bias-audit already carries."""
|
|
317
|
+
if x not in data or z not in data:
|
|
318
|
+
return f"unknown column(s); columns are {list(data)}"
|
|
319
|
+
rx = float(np.corrcoef(data[x], data[target])[0, 1])
|
|
320
|
+
rz = float(np.corrcoef(data[z], data[target])[0, 1])
|
|
321
|
+
product = np.asarray(data[x], float) * np.asarray(data[z], float)
|
|
322
|
+
rxz = float(np.corrcoef(product, data[target])[0, 1])
|
|
323
|
+
msg = (f"INTERACT: {x}*{z} vs {target} (n={len(data[target])}). "
|
|
324
|
+
f"Individually: corr({x},{target})={rx:+.2f}, corr({z},{target})={rz:+.2f}. "
|
|
325
|
+
f"Product term: corr({x}*{z},{target})={rxz:+.2f}.")
|
|
326
|
+
if abs(rxz) - max(abs(rx), abs(rz)) > 0.15:
|
|
327
|
+
msg += (f"\n[FOUND] the product explains far more than either variable alone -- "
|
|
328
|
+
f"a real candidate INTERACTION driver, invisible to CORR/ADJUST's linear-"
|
|
329
|
+
f"only view. Not yet a full causal claim: confirm neither {x} nor {z} is "
|
|
330
|
+
f"downstream of {target} before trusting this.")
|
|
331
|
+
else:
|
|
332
|
+
msg += "\nNo meaningful interaction signal beyond what the individual variables already show."
|
|
333
|
+
return msg
|
|
334
|
+
|
|
335
|
+
def refute(x, zs):
|
|
336
|
+
"""DoWhy-backed refutation testing -- a SECOND, independently-derived
|
|
337
|
+
robustness check on the SAME x|zs hypothesis ADJUST already tested,
|
|
338
|
+
using an established causal-inference library's estimator plus three
|
|
339
|
+
real perturbation tests, instead of this file's own hand-rolled
|
|
340
|
+
Cinelli-Hazlett RV. Does NOT resolve confounder-vs-mediator ambiguity
|
|
341
|
+
(see ADJUST's bias-audit for that, still a DAG/time-order question
|
|
342
|
+
data alone can't answer) -- it only asks whether the NUMERICAL
|
|
343
|
+
estimate survives real perturbation:
|
|
344
|
+
placebo treatment -- effect should COLLAPSE toward 0 (confirms
|
|
345
|
+
the estimate isn't a fitting-procedure
|
|
346
|
+
artifact; treatment is randomly permuted)
|
|
347
|
+
random common cause -- effect should barely CHANGE (a real,
|
|
348
|
+
already-adjusted effect shouldn't move much
|
|
349
|
+
from one more irrelevant confounder)
|
|
350
|
+
data subset (80%) -- effect should barely CHANGE (not driven by
|
|
351
|
+
a handful of influential points)
|
|
352
|
+
Stochastic (permutation/resampling-based) -- re-running can shift the
|
|
353
|
+
exact numbers slightly; the qualitative collapsed/stable read is what
|
|
354
|
+
matters, not the third decimal place."""
|
|
355
|
+
if not _dowhy_available():
|
|
356
|
+
return "REFUTE unavailable: pip install dowhy to enable (not a core dependency)."
|
|
357
|
+
if x not in data:
|
|
358
|
+
return f"unknown column '{x}'; columns are {list(data)}"
|
|
359
|
+
zs = [z for z in zs if z in data and z not in (x, target)]
|
|
360
|
+
import pandas as pd
|
|
361
|
+
from dowhy import CausalModel
|
|
362
|
+
df = pd.DataFrame({c: data[c] for c in data})
|
|
363
|
+
try:
|
|
364
|
+
cm = CausalModel(data=df, treatment=x, outcome=target, common_causes=zs or None)
|
|
365
|
+
identified = cm.identify_effect(proceed_when_unidentifiable=True)
|
|
366
|
+
est = cm.estimate_effect(identified, method_name="backdoor.linear_regression")
|
|
367
|
+
orig = float(est.value)
|
|
368
|
+
placebo = cm.refute_estimate(identified, est, method_name="placebo_treatment_refuter", placebo_type="permute")
|
|
369
|
+
rcc = cm.refute_estimate(identified, est, method_name="random_common_cause")
|
|
370
|
+
subset = cm.refute_estimate(identified, est, method_name="data_subset_refuter", subset_fraction=0.8)
|
|
371
|
+
except Exception as e:
|
|
372
|
+
return f"REFUTE error (DoWhy could not fit/refute this model): {e}"
|
|
373
|
+
|
|
374
|
+
zst = ", ".join(zs) if zs else "nothing"
|
|
375
|
+
p_new, rcc_new, subset_new = float(placebo.new_effect), float(rcc.new_effect), float(subset.new_effect)
|
|
376
|
+
scale = abs(orig) if abs(orig) > 1e-9 else 1e-9
|
|
377
|
+
collapsed = abs(p_new) < 0.1 * scale
|
|
378
|
+
stable_rcc = abs(rcc_new - orig) < 0.2 * scale
|
|
379
|
+
stable_subset = abs(subset_new - orig) < 0.2 * scale
|
|
380
|
+
n_pass = sum([collapsed, stable_rcc, stable_subset])
|
|
381
|
+
|
|
382
|
+
msg = (f"REFUTE (DoWhy): effect of {x} on {target} controlling for [{zst}] -- "
|
|
383
|
+
f"original estimate {orig:+.3f} (backdoor.linear_regression, independent of ADJUST's own fit).\n"
|
|
384
|
+
f" placebo treatment: new effect {p_new:+.3f} -- "
|
|
385
|
+
+ ("collapsed toward 0, as expected for a real effect." if collapsed
|
|
386
|
+
else "did NOT collapse -- suspicious; the estimate may reflect the fitting procedure, not a real relationship.") + "\n"
|
|
387
|
+
f" random common cause: new effect {rcc_new:+.3f} -- "
|
|
388
|
+
+ ("stable." if stable_rcc else "changed notably -- sensitive to an irrelevant confounder, a fragility signal.") + "\n"
|
|
389
|
+
f" data subset (80%): new effect {subset_new:+.3f} -- "
|
|
390
|
+
+ ("stable." if stable_subset else "changed notably -- may be driven by a subset of influential points.") + "\n"
|
|
391
|
+
f"[{n_pass}/3 refutation checks consistent with a real, stable effect]")
|
|
392
|
+
return msg
|
|
393
|
+
|
|
394
|
+
return corr, run, strat, adjust, interact, refute, iv
|
|
395
|
+
|
|
396
|
+
# ---------------- RECALL: optional semantic search over an external vault (OBSERVE) ----------------
|
|
397
|
+
_OBSERVE_PATH = os.environ.get("METHODLM_OBSERVE_PATH") # optional: path to an OBSERVE checkout
|
|
398
|
+
_VAULT_INDEX_DIR = os.environ.get("METHODLM_VAULT_INDEX") # optional: path to a pre-built OBSERVE index dir
|
|
399
|
+
_VAULT_MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
|
400
|
+
_VAULT_ENGINE = None # lazy singleton: loading the embedding model takes real seconds, do it once
|
|
401
|
+
|
|
402
|
+
def _vault_engine():
|
|
403
|
+
"""RECALL is fully optional, same pattern as the tritkit second witness above: unset
|
|
404
|
+
either env var and RECALL just reports itself unavailable rather than erroring. False
|
|
405
|
+
(not None) once tried-and-failed, so a missing index doesn't retry every call."""
|
|
406
|
+
global _VAULT_ENGINE
|
|
407
|
+
if _VAULT_ENGINE is not None:
|
|
408
|
+
return _VAULT_ENGINE
|
|
409
|
+
if not _OBSERVE_PATH or not _VAULT_INDEX_DIR:
|
|
410
|
+
_VAULT_ENGINE = False
|
|
411
|
+
return _VAULT_ENGINE
|
|
412
|
+
if _OBSERVE_PATH not in sys.path:
|
|
413
|
+
sys.path.insert(0, _OBSERVE_PATH)
|
|
414
|
+
try:
|
|
415
|
+
from search_engine import SearchEngine
|
|
416
|
+
eng = SearchEngine()
|
|
417
|
+
eng.load_blocking(_VAULT_INDEX_DIR, _VAULT_MODEL)
|
|
418
|
+
_VAULT_ENGINE = eng if eng.ready else False
|
|
419
|
+
except Exception as e:
|
|
420
|
+
print(f"[warn] RECALL vault search unavailable: {e}")
|
|
421
|
+
_VAULT_ENGINE = False
|
|
422
|
+
return _VAULT_ENGINE
|
|
423
|
+
|
|
424
|
+
def recall(query, k=5):
|
|
425
|
+
"""Real semantic search (MiniLM embeddings, not a keyword grep) over an external OBSERVE-
|
|
426
|
+
indexed corpus (e.g. a notes vault) -- 'did we already investigate/decide this before'
|
|
427
|
+
memory. NOT a causal test: does not count toward nrun and cannot satisfy the pre-FINAL
|
|
428
|
+
test requirement, same as ATTR/CORR. Fully optional -- see METHODLM_OBSERVE_PATH /
|
|
429
|
+
METHODLM_VAULT_INDEX above."""
|
|
430
|
+
eng = _vault_engine()
|
|
431
|
+
if not eng:
|
|
432
|
+
return ("RECALL unavailable: set METHODLM_OBSERVE_PATH (a checkout of OBSERVE, "
|
|
433
|
+
"https://github.com/tritsystem/observe-api) and METHODLM_VAULT_INDEX (a "
|
|
434
|
+
"pre-built semantic index directory) to enable it.")
|
|
435
|
+
hits = eng.search(query, k=k)
|
|
436
|
+
if not hits:
|
|
437
|
+
return f"RECALL: no vault matches for '{query}'."
|
|
438
|
+
lines = [f"RECALL: top {len(hits)} vault match(es) for '{query}':"]
|
|
439
|
+
for h in hits:
|
|
440
|
+
lines.append(f" [{h['score']:.2f}] {os.path.basename(h['path'])}: {h['preview'][:160].strip()}")
|
|
441
|
+
return "\n".join(lines)
|
|
442
|
+
|
|
443
|
+
def _split_confounders(s):
|
|
444
|
+
"""Split an ADJUST/REFUTE/IV confounder-list argument on comma AND/OR whitespace.
|
|
445
|
+
|
|
446
|
+
Real bug, found while benchmarking the contrastive-decoding backend (a Qwen2.5-1.5B/
|
|
447
|
+
0.5B pair): it reliably writes 'ADJUST: x | beta gamma delta epsilon' -- SPACE-separated,
|
|
448
|
+
no commas -- a plausible, common-enough LLM output style that the old `.split(",")` had
|
|
449
|
+
no way to handle: with no comma present the whole tail became ONE non-matching token,
|
|
450
|
+
got filtered out by `if z in data`, and every such ADJUST call silently ran as
|
|
451
|
+
'controlling for [nothing]' (a bare, unadjusted correlation) instead of the real
|
|
452
|
+
adjustment -- while the tool's own [FULL SET] reference line, immune to this because it
|
|
453
|
+
loops confounders individually, showed the correct number a few words later in the SAME
|
|
454
|
+
message. It also corrupted the loop's own proactive "all candidates tested, conclude
|
|
455
|
+
now" nudge, which regex-greps the FIRST 'RV=' in the tool text -- the inflated raw/
|
|
456
|
+
unadjusted RV, not the correct adjusted one. Splitting on comma OR whitespace fixes both
|
|
457
|
+
silently -- no behavior change for the (still valid) comma-separated style every other
|
|
458
|
+
backend has been using."""
|
|
459
|
+
return [z for z in re.split(r"[,\s]+", s.strip()) if z]
|
|
460
|
+
|
|
461
|
+
# ---------------- REASON: the copilot loop ----------------
|
|
462
|
+
def build_system(cols, target, interventional):
|
|
463
|
+
names = ", ".join(cols)
|
|
464
|
+
a, b = [c for c in cols if c != target][:2]
|
|
465
|
+
runline = (f" RUN: vary={a}, clamp={{{b}:50}} (true controlled experiment; clamp value is a NUMBER)"
|
|
466
|
+
if interventional else
|
|
467
|
+
f" STRAT: {a},{b} (quick check: corr({a},{target}) inside bands of {b})")
|
|
468
|
+
return textwrap.dedent(f"""\
|
|
469
|
+
You are a research-methodology copilot running a real instrument by the gbranaa-hue
|
|
470
|
+
method. A correlation is never a cause; suspect the boring explanation (artifact,
|
|
471
|
+
confound, the instrument itself) first; trust only tests that could have failed.
|
|
472
|
+
|
|
473
|
+
Data columns (EXACT names): {names}. Target: {target}.
|
|
474
|
+
Tools -- end each reply with exactly one tool line:
|
|
475
|
+
CORR: {a},{target}
|
|
476
|
+
{runline}
|
|
477
|
+
ADJUST: {a} | <other confounder columns> (backdoor adjustment + sensitivity: effect
|
|
478
|
+
of {a} on {target} controlling for the listed confounders,
|
|
479
|
+
with a robustness value + a collider/mediator bias audit)
|
|
480
|
+
IV: {a} | instrument=<column>[, controls=<other columns>] (2SLS: use when ADJUST's
|
|
481
|
+
robustness value is low or a confounder is UNMEASURED --
|
|
482
|
+
ADJUST can only control for confounders you actually have
|
|
483
|
+
as columns; IV can recover the true effect even with
|
|
484
|
+
unmeasured confounding IF <column> is a genuine instrument
|
|
485
|
+
that moves {a} but has NO direct effect on {target} except
|
|
486
|
+
through {a}. Reports the first-stage F-statistic (F<10 =
|
|
487
|
+
weak instrument, unreliable estimate) and states plainly
|
|
488
|
+
that the exclusion restriction cannot be verified from
|
|
489
|
+
data alone -- same honest boundary as ADJUST's collider/
|
|
490
|
+
mediator split.)
|
|
491
|
+
INTERACT: {a},<other column> (tests the PRODUCT of two columns as a driver --
|
|
492
|
+
CORR/ADJUST are linear and CANNOT see a pure interaction
|
|
493
|
+
effect, even a perfect one, no matter how strong; if every
|
|
494
|
+
candidate looks fragile alone under ADJUST, try this before
|
|
495
|
+
concluding "no driver")
|
|
496
|
+
REFUTE: {a} | <same confounder columns as your ADJUST call> (SECOND, independent
|
|
497
|
+
robustness check on a candidate ADJUST already found
|
|
498
|
+
promising -- DoWhy's own estimator + 3 real perturbation
|
|
499
|
+
tests: placebo treatment should COLLAPSE the effect,
|
|
500
|
+
random common cause and data-subset should barely change
|
|
501
|
+
it. Use AFTER ADJUST on the same candidate, not instead of
|
|
502
|
+
it -- ADJUST's bias-audit and REFUTE's perturbation checks
|
|
503
|
+
catch different failure modes.)
|
|
504
|
+
ATTR: (the learned ternary model's evidence per column)
|
|
505
|
+
RECALL: <free-text query> (semantic search over past vault notes/decisions for
|
|
506
|
+
relevant prior work -- memory, NOT a causal test; use it
|
|
507
|
+
ONCE, if at all, to check "did we already find this" before
|
|
508
|
+
re-deriving -- it cannot be used a second time; unavailable
|
|
509
|
+
unless METHODLM_OBSERVE_PATH/METHODLM_VAULT_INDEX are set)
|
|
510
|
+
HOW TO FIND THE DRIVER: for a candidate X, run 'ADJUST: X | <the other candidate
|
|
511
|
+
columns>'. The tool shows the full-set result AND a [BIAS-AUDIT]. HEED IT: drop any
|
|
512
|
+
variable flagged as a COLLIDER (conditioning on it manufactures a fake effect), and do
|
|
513
|
+
NOT adjust for a MEDIATOR (a variable on the X->{target} path or measured after X) --
|
|
514
|
+
'adjust for everything' is the Table 2 fallacy. Condition on prior common causes only.
|
|
515
|
+
If X's adjusted partial corr stays large with a high robustness value, X drives
|
|
516
|
+
{target}; if it COLLAPSES toward 0 (or RV < 0.10), X is a confounded bystander.
|
|
517
|
+
CRITICAL: the driver is the candidate that SURVIVES full adjustment -- it is NEVER
|
|
518
|
+
the variable you controlled for. Do not name a conditioning/control variable as the
|
|
519
|
+
cause; that is backwards. Strategy: ADJUST the tempting/decoy variable
|
|
520
|
+
first; if it collapses, ADJUST the other strong candidate to confirm the real
|
|
521
|
+
driver; once one survives, REFUTE it on the same X | <same confounders> as a second,
|
|
522
|
+
independently-derived check before concluding (if REFUTE is unavailable it will say so
|
|
523
|
+
-- don't let that block FINAL, ADJUST's own robustness value already counts as a test).
|
|
524
|
+
Before any ADJUST/STRAT/RUN/INTERACT/REFUTE/IV, include a 'PREREGISTER:' line in the
|
|
525
|
+
SAME reply naming the same X you test and what result confirms vs disconfirms. Comparing
|
|
526
|
+
two correlations is NOT a test. You MUST run at least one ADJUST/STRAT/RUN/INTERACT/IV
|
|
527
|
+
(never only CORR) before any FINAL. Your FINAL must name the variable (or product of two
|
|
528
|
+
variables) whose effect SURVIVED as the driver (or say none did). Be brief.""")
|
|
529
|
+
|
|
530
|
+
def ask(system, messages, n=230):
|
|
531
|
+
return backend().generate(system, messages, n)
|
|
532
|
+
|
|
533
|
+
def vanilla_answer(question, n=200):
|
|
534
|
+
"""The SAME model, no method and no tools -- a plain analyst. The control racer."""
|
|
535
|
+
sysp = ("You are a helpful data analyst. Answer the question directly and concisely: "
|
|
536
|
+
"give your conclusion and a recommendation.")
|
|
537
|
+
return ask(sysp, [{"role": "user", "content": question}], n)
|
|
538
|
+
|
|
539
|
+
def investigate(name, data, target, question, interventional, answer_key=None, ingest_report=None):
|
|
540
|
+
ledger = os.path.join(HERE, f"ledger_{name}.txt")
|
|
541
|
+
led = open(ledger, "w", encoding="utf-8")
|
|
542
|
+
def out(s):
|
|
543
|
+
led.write(s + "\n"); led.flush()
|
|
544
|
+
try: print(s, flush=True)
|
|
545
|
+
except UnicodeEncodeError: print(s.encode("ascii", "replace").decode(), flush=True)
|
|
546
|
+
|
|
547
|
+
if ingest_report:
|
|
548
|
+
out(f"[ingest]\n{ingest_report}\n")
|
|
549
|
+
t0 = time.time()
|
|
550
|
+
try:
|
|
551
|
+
nmse, share, gate = train_readout(data, target)
|
|
552
|
+
top = ", ".join(f"{k} {v*100:.0f}%" for k, v in list(share.items())[:4])
|
|
553
|
+
out(f"[compute] ternary readout NMSE {nmse:.2f} | evidence: {top} | gate: {gate}")
|
|
554
|
+
except ImportError:
|
|
555
|
+
nmse, share, gate = None, {}, []
|
|
556
|
+
out("[compute] ternary second witness skipped (set METHODLM_TRITKIT + install torch to enable)")
|
|
557
|
+
corr, run, strat, adjust, interact, refute, iv = make_tools(data, target, interventional)
|
|
558
|
+
system = build_system(list(data), target, interventional)
|
|
559
|
+
|
|
560
|
+
msgs = [{"role": "user", "content": question}]
|
|
561
|
+
pre = nrun = n_recall = 0
|
|
562
|
+
verdict = "(no verdict — ran out of turns)"
|
|
563
|
+
seen = {} # loop guard: signatures of tests already run
|
|
564
|
+
stuck = 0 # consecutive turns producing no usable tool call
|
|
565
|
+
best_rv = {} # best (highest) RV seen per tested column -- drives the proactive
|
|
566
|
+
# "all candidates tested, conclude now" nudge
|
|
567
|
+
# 12 was the original cap. Real bug, found by rerunning a genuine 8-candidate financial
|
|
568
|
+
# panel (copper_xlk) after fixing the local-model context-window truncation below: the
|
|
569
|
+
# model correctly PREREGISTERed and ADJUSTed all 8 candidates, correctly identified the
|
|
570
|
+
# one survivor (spy_return, RV=0.10) and got the "[ALL CANDIDATES TESTED] ... reply FINAL
|
|
571
|
+
# now" nudge on turn 12 itself -- but turn 12 was the LAST turn in range(1,13), so there
|
|
572
|
+
# was no turn left to actually emit FINAL and the run still ended "(no verdict -- ran out
|
|
573
|
+
# of turns)" despite the correct answer already sitting in the transcript. A panel with
|
|
574
|
+
# N candidate columns needs >=N ADJUST turns plus headroom for PREREGISTER-missing/syntax
|
|
575
|
+
# retries (both real, observed failure modes) plus 1+ for FINAL itself -- 12 is too tight
|
|
576
|
+
# for anything past ~6-7 candidates. 24 gives a ~10-12 candidate panel real room without
|
|
577
|
+
# being unbounded (the loop-guard/no-progress-guard above still force an early stop on a
|
|
578
|
+
# genuinely stuck run).
|
|
579
|
+
for turn in range(1, 25):
|
|
580
|
+
raw = ask(system, msgs)
|
|
581
|
+
# ENFORCE ONE ACTION PER TURN: keep text up to and including the first tool
|
|
582
|
+
# directive (or FINAL); drop anything the model dumped after it.
|
|
583
|
+
cut = None
|
|
584
|
+
for m in re.finditer(r"^\s*(CORR|RUN|STRAT|ADJUST|INTERACT|REFUTE|IV|ATTR|RECALL|FINAL)\b.*$", raw, re.MULTILINE | re.IGNORECASE):
|
|
585
|
+
cut = m.end(); break
|
|
586
|
+
reply = raw[:cut] if cut else raw
|
|
587
|
+
out(f"\n--- copilot turn {turn} ---\n{reply}")
|
|
588
|
+
msgs.append({"role": "assistant", "content": reply})
|
|
589
|
+
if "PREREGISTER:" in reply: pre += 1
|
|
590
|
+
m_run = re.search(r"RUN:\s*vary=(\w+),\s*clamp=\{([^}]*)\}", reply)
|
|
591
|
+
m_adj = re.search(r"ADJUST:\s*(\w+)\s*\|\s*([\w,\s]*)", reply)
|
|
592
|
+
m_str = re.search(r"STRAT:\s*(\w+)\s*,\s*(\w+)", reply)
|
|
593
|
+
m_cor = re.search(r"CORR:\s*(\w+)\s*,\s*(\w+)", reply)
|
|
594
|
+
m_attr = re.search(r"\bATTR:", reply)
|
|
595
|
+
m_rec = re.search(r"RECALL:\s*(.+)$", reply, re.MULTILINE)
|
|
596
|
+
m_int = re.search(r"INTERACT:\s*(\w+)\s*,\s*(\w+)", reply)
|
|
597
|
+
m_ref = re.search(r"REFUTE:\s*(\w+)\s*\|\s*([\w,\s]*)", reply)
|
|
598
|
+
m_iv = re.search(r"IV:\s*(\w+)\s*\|\s*instrument\s*=\s*(\w+)(?:\s*,\s*controls\s*=\s*([\w,\s]*))?", reply)
|
|
599
|
+
# FINAL honored ONLY when it's not a hedge AND a real test has run
|
|
600
|
+
if re.search(r"\bfinal\s*:", reply, re.IGNORECASE) and not (m_run or m_adj or m_str or m_cor or m_attr or m_int or m_ref or m_iv):
|
|
601
|
+
if nrun < 1:
|
|
602
|
+
out("[TOOL] REFUSED: comparing correlations is not a test. Run one STRAT or "
|
|
603
|
+
"RUN before concluding.")
|
|
604
|
+
msgs.append({"role": "user", "content": "REFUSED: run at least one STRAT or "
|
|
605
|
+
"RUN test before FINAL."})
|
|
606
|
+
continue
|
|
607
|
+
verdict = re.sub(r"^.*?final\s*:", "", reply, flags=re.IGNORECASE | re.DOTALL).strip()
|
|
608
|
+
out("\n[done] verdict reached."); break
|
|
609
|
+
if m_run:
|
|
610
|
+
clamp = {k: float(v) for k, v in re.findall(r"(\w+)\s*:\s*([-\d.]+)", m_run.group(2))}
|
|
611
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
612
|
+
else "Clamp needs numbers, e.g. clamp={humidity:50}" if not clamp
|
|
613
|
+
else run(m_run.group(1), clamp)); nrun += m_run and bool(clamp) and "PREREGISTER:" in reply
|
|
614
|
+
elif m_adj:
|
|
615
|
+
zs = _split_confounders(m_adj.group(2))
|
|
616
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
617
|
+
else adjust(m_adj.group(1), zs)); nrun += "PREREGISTER:" in reply
|
|
618
|
+
elif m_str:
|
|
619
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
620
|
+
else strat(m_str.group(1), m_str.group(2))); nrun += bool(m_str) and "PREREGISTER:" in reply
|
|
621
|
+
elif m_cor:
|
|
622
|
+
res = corr(m_cor.group(1), m_cor.group(2))
|
|
623
|
+
elif m_int:
|
|
624
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
625
|
+
else interact(m_int.group(1), m_int.group(2))); nrun += "PREREGISTER:" in reply
|
|
626
|
+
elif m_ref:
|
|
627
|
+
zs = _split_confounders(m_ref.group(2))
|
|
628
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
629
|
+
else refute(m_ref.group(1), zs)); nrun += "PREREGISTER:" in reply
|
|
630
|
+
elif m_iv:
|
|
631
|
+
ctrls = _split_confounders(m_iv.group(3) or "")
|
|
632
|
+
res = ("REFUSED: PREREGISTER in the same reply first." if "PREREGISTER:" not in reply
|
|
633
|
+
else iv(m_iv.group(1), m_iv.group(2), ctrls)); nrun += "PREREGISTER:" in reply
|
|
634
|
+
elif re.search(r"\bATTR:", reply):
|
|
635
|
+
if nmse is None:
|
|
636
|
+
res = "Ternary second witness unavailable (install torch + set METHODLM_TRITKIT)."
|
|
637
|
+
else:
|
|
638
|
+
res = (f"Ternary readout (NMSE {nmse:.2f}) evidence share: "
|
|
639
|
+
+ ", ".join(f"{k}: {v*100:.0f}%" for k, v in share.items())
|
|
640
|
+
+ f". Gate connects: {', '.join(gate)}.")
|
|
641
|
+
elif m_rec:
|
|
642
|
+
# SINGLE-USE GUARD: a weak model echoing the [TOOL] result text back as its own
|
|
643
|
+
# next RECALL: line was a real, observed failure mode (spiraling into nested
|
|
644
|
+
# self-quoting, e.g. "RECALL: top 5 vault match(es) for 'top 5 vault match(es)...'"
|
|
645
|
+
# until it lost the ability to emit any tool line at all). RECALL's job is a
|
|
646
|
+
# one-time "did we already do this" memory check, not a repeatable action, so cap
|
|
647
|
+
# it at one real call per investigation regardless of query content.
|
|
648
|
+
n_recall += 1
|
|
649
|
+
if n_recall > 1:
|
|
650
|
+
res = ("REFUSED: RECALL already used once this investigation -- it is a one-time "
|
|
651
|
+
"memory check, not repeatable. PREREGISTER and run a real causal test "
|
|
652
|
+
"(ADJUST/STRAT/RUN) now.")
|
|
653
|
+
else:
|
|
654
|
+
query = m_rec.group(1).strip()[:200] # cap length: reject echoed-tool-output blowup
|
|
655
|
+
res = recall(query)
|
|
656
|
+
else:
|
|
657
|
+
# Real observed failure mode (live UI test): the model wrote free prose containing
|
|
658
|
+
# "ADJUST" instead of the exact syntax ("ADJUST: effect of humidity on error
|
|
659
|
+
# controlling for [temperature, vibration]...") and got a generic "No tool
|
|
660
|
+
# recognized" that didn't teach it the fix -- it then burned the rest of its turn
|
|
661
|
+
# budget without ever recovering. Detecting the attempted keyword and echoing back
|
|
662
|
+
# the exact expected syntax gives it a real shot at self-correcting next turn.
|
|
663
|
+
upper = reply.upper()
|
|
664
|
+
if "REFUTE" in upper:
|
|
665
|
+
hint = " Correct REFUTE syntax: 'REFUTE: x | z1,z2' (same shape as ADJUST -- pipe before confounders)."
|
|
666
|
+
elif "ADJUST" in upper:
|
|
667
|
+
hint = " Correct ADJUST syntax: 'ADJUST: x | z1,z2' (pipe before confounders, comma-separated, no prose)."
|
|
668
|
+
elif "IV:" in upper:
|
|
669
|
+
# "IV" alone is too short to test as a bare substring (matches inside DRIVE,
|
|
670
|
+
# SURVIVE, POSITIVE, etc. in ordinary prose) -- require the colon the real
|
|
671
|
+
# syntax always has.
|
|
672
|
+
hint = (" Correct IV syntax: 'IV: x | instrument=z' (optionally add "
|
|
673
|
+
"', controls=c1,c2'); pipe before instrument=, same shape as ADJUST.")
|
|
674
|
+
elif "INTERACT" in upper:
|
|
675
|
+
hint = " Correct INTERACT syntax: 'INTERACT: x,z' (exactly two column names, comma-separated)."
|
|
676
|
+
elif "STRAT" in upper:
|
|
677
|
+
hint = " Correct STRAT syntax: 'STRAT: x,z' (exactly two column names, comma-separated)."
|
|
678
|
+
elif "RUN" in upper:
|
|
679
|
+
hint = " Correct RUN syntax: 'RUN: vary=x, clamp={z:NUMBER}' (clamp value must be a number)."
|
|
680
|
+
elif "CORR" in upper:
|
|
681
|
+
hint = " Correct CORR syntax: 'CORR: x,y' (exactly two column names, comma-separated)."
|
|
682
|
+
else:
|
|
683
|
+
hint = ""
|
|
684
|
+
res = f"No tool recognized.{hint} Use CORR:, STRAT:/RUN:, ADJUST:, IV:, INTERACT:, REFUTE:, ATTR:, RECALL:, or FINAL:."
|
|
685
|
+
# LOOP GUARD: a weak model can re-run the same test forever. If a real test
|
|
686
|
+
# repeats, don't re-run it -- nudge to conclude; force a stop on a 2nd repeat.
|
|
687
|
+
# Tagged with the tool name: REFUTE is DESIGNED to be re-run on the exact same
|
|
688
|
+
# (x, zs) an ADJUST call already used (a second, independent robustness check
|
|
689
|
+
# on the same candidate) -- an untagged sig would wrongly flag that as a repeat.
|
|
690
|
+
_sig_src = [("RUN", m_run), ("ADJUST", m_adj), ("STRAT", m_str), ("CORR", m_cor),
|
|
691
|
+
("INTERACT", m_int), ("REFUTE", m_ref), ("IV", m_iv)]
|
|
692
|
+
sig = next(((tag, g.groups()) for tag, g in _sig_src if g), None)
|
|
693
|
+
if sig and not str(res).startswith(("REFUSED", "Clamp")):
|
|
694
|
+
# Track the best (highest) RV seen per tested column -- real bug, found by
|
|
695
|
+
# running this exact scenario: the model found the correct driver (humidity,
|
|
696
|
+
# RV=0.42) at turn 6, ruled out the only other candidate at turn 7, then instead
|
|
697
|
+
# of concluding, re-ran an already-collapsed test a 3rd time and got hard-stopped
|
|
698
|
+
# by the loop-guard with NO verdict, despite the right answer already sitting in
|
|
699
|
+
# the transcript. The "all candidates tested, none left" signal already existed
|
|
700
|
+
# but only fired reactively inside the repeat-branch below -- one turn too late
|
|
701
|
+
# to have prevented the wasted turn-8 retry. Checking it proactively, right after
|
|
702
|
+
# the test that completes coverage, gives the model the strongest signal a full
|
|
703
|
+
# turn earlier. sig[1][0] (not sig[0]) -- sig is (tag, groups) here, tagged so
|
|
704
|
+
# REFUTE re-testing the same candidate ADJUST already used isn't misread as a repeat.
|
|
705
|
+
rv_match = re.search(r"RV=([\d.]+)", str(res))
|
|
706
|
+
if rv_match:
|
|
707
|
+
col = sig[1][0]
|
|
708
|
+
best_rv[col] = max(best_rv.get(col, 0.0), float(rv_match.group(1)))
|
|
709
|
+
all_candidates = [c for c in data if c != target]
|
|
710
|
+
if set(best_rv) >= set(all_candidates):
|
|
711
|
+
survivors = [c for c, rv in best_rv.items() if rv >= 0.10]
|
|
712
|
+
bystanders = [c for c in all_candidates if c not in survivors]
|
|
713
|
+
if survivors:
|
|
714
|
+
res = (f"{res}\n[ALL CANDIDATES TESTED] Survivor(s) with RV>=0.10: "
|
|
715
|
+
f"{', '.join(survivors)}. Bystander(s) with RV<0.10: "
|
|
716
|
+
f"{', '.join(bystanders) or 'none'}. You have enough evidence -- "
|
|
717
|
+
f"reply FINAL now naming the survivor as the driver. Do not run "
|
|
718
|
+
f"another test.")
|
|
719
|
+
if sig in seen:
|
|
720
|
+
seen[sig] += 1
|
|
721
|
+
# nudge toward the NEXT untested candidate (not FINAL) -- a collapsed test
|
|
722
|
+
# means that variable is a bystander, not that the job is done.
|
|
723
|
+
tested = ", ".join(sorted({s[1][0] for s in seen})) or "none"
|
|
724
|
+
untested = [c for c in data if c not in (target,) and c not in {s[1][0] for s in seen}]
|
|
725
|
+
res = (f"{res}\n[REPEAT: you already ran this. A collapsed effect (RV<0.10) means "
|
|
726
|
+
f"that variable is a BYSTANDER, not the answer. You have tested: {tested}. "
|
|
727
|
+
f"Now run ADJUST on a DIFFERENT untested candidate ({', '.join(untested) or 'none left'}) "
|
|
728
|
+
"to find the real driver. Reply FINAL only once a variable SURVIVES (high RV).]")
|
|
729
|
+
if seen[sig] >= 3:
|
|
730
|
+
out(f"[TOOL] {res}")
|
|
731
|
+
verdict = "(loop-guard: model repeated the same test without concluding)"
|
|
732
|
+
out("\n[done] loop-guard stopped a repeat loop."); break
|
|
733
|
+
else:
|
|
734
|
+
seen[sig] = 1
|
|
735
|
+
out(f"[TOOL] {res}")
|
|
736
|
+
# NO-PROGRESS guard: a model that can't emit tool syntax loops uselessly.
|
|
737
|
+
if str(res).startswith(("No tool recognized", "REFUSED")):
|
|
738
|
+
stuck += 1
|
|
739
|
+
if stuck >= 3:
|
|
740
|
+
verdict = "(model could not drive the tool protocol)"
|
|
741
|
+
out("\n[done] no-progress guard: model never produced a usable test."); break
|
|
742
|
+
else:
|
|
743
|
+
stuck = 0
|
|
744
|
+
msgs.append({"role": "user", "content": res})
|
|
745
|
+
if answer_key:
|
|
746
|
+
out(f"\n[ANSWER KEY -- the instrument was never told] {answer_key}")
|
|
747
|
+
out(f"\n[ledger] preregistrations {pre}, registered tests {nrun}, "
|
|
748
|
+
f"{time.time()-t0:.0f}s -> {ledger}")
|
|
749
|
+
led.close()
|
|
750
|
+
return {"verdict": verdict, "pre": pre, "nrun": nrun,
|
|
751
|
+
"nmse": nmse, "gate": gate, "share": share}
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
def finish_race(question, res, answer_key=None):
|
|
755
|
+
"""Same question, two racers: MethodLM (method + tools + tests) vs plain LLM."""
|
|
756
|
+
print("\n" + "=" * 62)
|
|
757
|
+
print("HEAD-TO-HEAD -- same question, two racers")
|
|
758
|
+
print("=" * 62)
|
|
759
|
+
v = vanilla_answer(question)
|
|
760
|
+
tested = f"{res['nrun']} test(s), {res['pre']} pre-reg" if res["nrun"] else "NO TEST RUN"
|
|
761
|
+
print(f"\n[ vanilla 3B | no method, no tools ]\n {v}\n")
|
|
762
|
+
print(f"[ MethodLM | method + tools | {tested} ]\n {res['verdict']}\n")
|
|
763
|
+
if answer_key:
|
|
764
|
+
print(f"[ answer key ] {answer_key}")
|
|
765
|
+
|
|
766
|
+
def main():
|
|
767
|
+
ap = argparse.ArgumentParser()
|
|
768
|
+
ap.add_argument("--demo", action="store_true")
|
|
769
|
+
ap.add_argument("--diabetes", action="store_true")
|
|
770
|
+
ap.add_argument("--data", help="any file: csv/tsv/json/jsonl/parquet/sqlite/npz/xlsx")
|
|
771
|
+
ap.add_argument("--csv", help="alias for --data")
|
|
772
|
+
ap.add_argument("--folder", help="labelled folder of text/image files")
|
|
773
|
+
ap.add_argument("--target")
|
|
774
|
+
ap.add_argument("--table", help="sqlite table name (optional)")
|
|
775
|
+
ap.add_argument("--query", help="sqlite SQL query (optional)")
|
|
776
|
+
ap.add_argument("--race", action="store_true", help="also run a plain-LLM racer on the same question")
|
|
777
|
+
ap.add_argument("--model", default="local", help="reasoning backend: local (Qwen-3B) | claude")
|
|
778
|
+
args = ap.parse_args()
|
|
779
|
+
global BACKEND
|
|
780
|
+
BACKEND = methodlm_models.get_model(args.model, HERE)
|
|
781
|
+
print(f"[methodlm] reasoning backend: {BACKEND.label}")
|
|
782
|
+
if args.demo:
|
|
783
|
+
data = demo_world()
|
|
784
|
+
r = float(np.corrcoef(data['temperature'], data['error'])[0, 1])
|
|
785
|
+
q = (f"Our sensor's error correlates with temperature (r={r:+.2f}) in 2000 logged "
|
|
786
|
+
"samples; the team wants to install cooling. Find what actually drives the "
|
|
787
|
+
"error before we spend the money.")
|
|
788
|
+
key = "error = f(humidity); temperature only co-rises with humidity via season."
|
|
789
|
+
res = investigate("demo", data, "error", q, True, key)
|
|
790
|
+
if args.race: finish_race(q, res, key)
|
|
791
|
+
elif args.diabetes:
|
|
792
|
+
data = load_diabetes()
|
|
793
|
+
r = float(np.corrcoef(data['bmi'], data['progression'])[0, 1])
|
|
794
|
+
q = (f"In 442 real diabetes patients, bmi correlates with one-year disease "
|
|
795
|
+
f"progression (r={r:+.2f}). The clinic wants to fund a weight-loss-only "
|
|
796
|
+
"program. Before they do: is bmi's link robust, or does it run through blood "
|
|
797
|
+
"serum markers like ltg? You cannot rerun patients; use STRAT.")
|
|
798
|
+
res = investigate("diabetes", data, "progression", q, False)
|
|
799
|
+
if args.race: finish_race(q, res)
|
|
800
|
+
elif args.folder and args.target:
|
|
801
|
+
from methodlm_io import load_folder, featurize, format_report
|
|
802
|
+
raw, notes = load_folder(args.folder)
|
|
803
|
+
data, rep = featurize(raw, args.target)
|
|
804
|
+
report = format_report(notes, rep, args.target)
|
|
805
|
+
cols = [c for c in data if c != args.target]
|
|
806
|
+
q = (f"Investigate what actually drives {args.target} across these files "
|
|
807
|
+
f"(features: {', '.join(cols[:12])}). Do not trust raw correlations.")
|
|
808
|
+
res = investigate(os.path.basename(args.folder.rstrip('/\\')) or "folder",
|
|
809
|
+
data, args.target, q, False, ingest_report=report)
|
|
810
|
+
if args.race: finish_race(q, res)
|
|
811
|
+
elif args.data or args.csv:
|
|
812
|
+
from methodlm_io import validate, load_any, featurize, format_report
|
|
813
|
+
path = args.data or args.csv
|
|
814
|
+
v = validate(path, args.target, table=args.table, query=args.query)
|
|
815
|
+
if not v["ok"]:
|
|
816
|
+
print("MethodLM cannot run yet:")
|
|
817
|
+
for e in v["errors"]:
|
|
818
|
+
print(f" x {e}")
|
|
819
|
+
if v["suggestions"]:
|
|
820
|
+
print(f" -> try: {', '.join(map(str, v['suggestions']))}")
|
|
821
|
+
if v["info"]:
|
|
822
|
+
print(" columns available: "
|
|
823
|
+
+ ", ".join(f"{c['name']} [{c['kind']}]" for c in v["info"]["columns"]))
|
|
824
|
+
return
|
|
825
|
+
raw, notes = load_any(path, table=args.table, query=args.query)
|
|
826
|
+
data, rep = featurize(raw, args.target)
|
|
827
|
+
report = format_report(notes, rep, args.target)
|
|
828
|
+
cols = [c for c in data if c != args.target]
|
|
829
|
+
q = (f"Investigate what actually drives {args.target} in this recorded dataset "
|
|
830
|
+
f"(columns: {', '.join(cols[:12])}{'...' if len(cols) > 12 else ''}). "
|
|
831
|
+
"Do not trust raw correlations.")
|
|
832
|
+
name = os.path.splitext(os.path.basename(path))[0]
|
|
833
|
+
res = investigate(name, data, args.target, q, False, ingest_report=report)
|
|
834
|
+
if args.race: finish_race(q, res)
|
|
835
|
+
else:
|
|
836
|
+
ap.print_help()
|
|
837
|
+
|
|
838
|
+
if __name__ == "__main__":
|
|
839
|
+
main()
|