strikeone 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- strikeone/__init__.py +3 -0
- strikeone/ai/__init__.py +7 -0
- strikeone/ai/aiconfig.py +108 -0
- strikeone/ai/commands.py +77 -0
- strikeone/ai/evidence.py +315 -0
- strikeone/ai/providers.py +152 -0
- strikeone/ai/validator.py +151 -0
- strikeone/audit.py +456 -0
- strikeone/cli.py +381 -0
- strikeone/config.py +48 -0
- strikeone/contract.py +301 -0
- strikeone/data.py +74 -0
- strikeone/entity.py +193 -0
- strikeone/episodes.py +157 -0
- strikeone/examples.py +112 -0
- strikeone/features.py +86 -0
- strikeone/metrics.py +240 -0
- strikeone/policy_engine.py +166 -0
- strikeone/route.py +118 -0
- strikeone/rpc.py +242 -0
- strikeone/seal.py +96 -0
- strikeone-1.0.0.dist-info/METADATA +606 -0
- strikeone-1.0.0.dist-info/RECORD +26 -0
- strikeone-1.0.0.dist-info/WHEEL +4 -0
- strikeone-1.0.0.dist-info/entry_points.txt +2 -0
- strikeone-1.0.0.dist-info/licenses/LICENSE +201 -0
strikeone/__init__.py
ADDED
strikeone/ai/__init__.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""strikeone.ai — the narration layer. AI is disabled by default; nothing
|
|
2
|
+
in this package is imported by the deterministic commands. The LLM never
|
|
3
|
+
computes or alters risk, selects thresholds, chooses actions, touches the
|
|
4
|
+
holdout, modifies any metric, or produces any number that appears on a
|
|
5
|
+
judging slide: it receives a finished evidence contract and returns prose
|
|
6
|
+
whose every factual claim is re-checked against that contract before
|
|
7
|
+
printing (strikeone.ai.validator)."""
|
strikeone/ai/aiconfig.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""AI-layer configuration: .strikeone-ai.toml — items 5 and 7.
|
|
2
|
+
|
|
3
|
+
AI is DISABLED by default: no file, no provider, and every deterministic
|
|
4
|
+
command behaves exactly as it does today. The file stores provider,
|
|
5
|
+
base_url, model and the NAME of the credential env var. It never stores a
|
|
6
|
+
secret: the writer refuses to persist any value that equals the value of
|
|
7
|
+
an environment variable whose name matches *KEY* or *TOKEN* (asserted by
|
|
8
|
+
test), and `strikeone ai setup` only DETECTS env vars — it never asks
|
|
9
|
+
for one (a masked prompt still leaves the key in scrollback and any
|
|
10
|
+
recording buffer).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
AI_CONFIG_FILE = ".strikeone-ai.toml"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class CredentialLeakError(RuntimeError):
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _secret_values() -> set:
|
|
27
|
+
vals = set()
|
|
28
|
+
for name, val in os.environ.items():
|
|
29
|
+
up = name.upper()
|
|
30
|
+
if ("KEY" in up or "TOKEN" in up) and val and len(val) >= 8:
|
|
31
|
+
vals.add(val)
|
|
32
|
+
return vals
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def guarded_write(path: Path | str, pairs: dict) -> None:
|
|
36
|
+
"""The ONLY writer of the AI config. Refuses to persist secrets."""
|
|
37
|
+
secrets = _secret_values()
|
|
38
|
+
for k, v in pairs.items():
|
|
39
|
+
if str(v) in secrets:
|
|
40
|
+
raise CredentialLeakError(
|
|
41
|
+
f"refusing to write {k!r}: its value matches a *KEY*/*TOKEN* "
|
|
42
|
+
"environment variable. Credentials stay in the environment; "
|
|
43
|
+
"the config stores only the env var's NAME.")
|
|
44
|
+
if k.lower() in ("api_key", "apikey", "token", "secret"):
|
|
45
|
+
raise CredentialLeakError(
|
|
46
|
+
f"refusing to write a field named {k!r}; store the env var "
|
|
47
|
+
"NAME under api_key_env instead.")
|
|
48
|
+
lines = ["[ai]"] + [f'{k} = "{v}"' for k, v in pairs.items()]
|
|
49
|
+
Path(path).write_text("\n".join(lines) + "\n")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class AIConfig:
|
|
54
|
+
provider: str = "" # "ollama" | "openai-compatible"
|
|
55
|
+
model: str = ""
|
|
56
|
+
base_url: str = ""
|
|
57
|
+
api_key_env: str = "OPENAI_API_KEY"
|
|
58
|
+
think: str = "" # "" | "on" | "off" (ollama hybrid reasoners)
|
|
59
|
+
|
|
60
|
+
@staticmethod
|
|
61
|
+
def load(path: Path | str = AI_CONFIG_FILE) -> "AIConfig | None":
|
|
62
|
+
p = Path(path)
|
|
63
|
+
if not p.exists():
|
|
64
|
+
return None
|
|
65
|
+
import tomllib
|
|
66
|
+
raw = tomllib.loads(p.read_text()).get("ai", {})
|
|
67
|
+
if not raw.get("provider"):
|
|
68
|
+
return None
|
|
69
|
+
return AIConfig(provider=raw.get("provider", ""),
|
|
70
|
+
model=raw.get("model", ""),
|
|
71
|
+
base_url=raw.get("base_url", ""),
|
|
72
|
+
api_key_env=raw.get("api_key_env", "OPENAI_API_KEY"),
|
|
73
|
+
think=raw.get("think", ""))
|
|
74
|
+
|
|
75
|
+
def save(self, path: Path | str = AI_CONFIG_FILE) -> None:
|
|
76
|
+
pairs = {"provider": self.provider, "model": self.model}
|
|
77
|
+
if self.base_url:
|
|
78
|
+
pairs["base_url"] = self.base_url
|
|
79
|
+
if self.think:
|
|
80
|
+
pairs["think"] = self.think
|
|
81
|
+
if self.provider == "openai-compatible":
|
|
82
|
+
pairs["api_key_env"] = self.api_key_env
|
|
83
|
+
guarded_write(path, pairs)
|
|
84
|
+
|
|
85
|
+
def build(self):
|
|
86
|
+
from strikeone.ai.providers import (OllamaProvider,
|
|
87
|
+
OpenAICompatibleProvider)
|
|
88
|
+
if self.provider == "ollama":
|
|
89
|
+
think = {"on": True, "off": False}.get(self.think)
|
|
90
|
+
return OllamaProvider(model=self.model,
|
|
91
|
+
base_url=self.base_url
|
|
92
|
+
or "http://localhost:11434",
|
|
93
|
+
think=think)
|
|
94
|
+
if self.provider == "openai-compatible":
|
|
95
|
+
if not self.base_url:
|
|
96
|
+
raise ValueError("openai-compatible needs base_url")
|
|
97
|
+
return OpenAICompatibleProvider(model=self.model,
|
|
98
|
+
base_url=self.base_url,
|
|
99
|
+
api_key_env=self.api_key_env)
|
|
100
|
+
raise ValueError(f"unknown provider {self.provider!r}")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
KNOWN_KEY_ENVS = ["OPENAI_API_KEY", "OPENROUTER_API_KEY"]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def detect_env() -> list:
|
|
107
|
+
"""Names (never values) of known credential env vars that are set."""
|
|
108
|
+
return [n for n in KNOWN_KEY_ENVS if os.environ.get(n)]
|
strikeone/ai/commands.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""The three AI commands — item 4. why / timeline / compare. No more.
|
|
2
|
+
|
|
3
|
+
Deliberately NOT built, so nobody has to wonder: /challenge (invites the
|
|
4
|
+
model to second-guess a deterministic decision — it has no calibration
|
|
5
|
+
and reads as an unmeasured second fraud model), /investigate, /simulate,
|
|
6
|
+
/metrics (duplicates audit), hybrid auto-escalation, provider menus, our
|
|
7
|
+
own inference server.
|
|
8
|
+
|
|
9
|
+
Pipeline, in this order and only this order:
|
|
10
|
+
CLI → intent parser (argparse) → deterministic router
|
|
11
|
+
(evidence.BUILDERS) → engine computes the evidence contract →
|
|
12
|
+
provider narrates → citation validator re-checks every claim →
|
|
13
|
+
validated text is printed. The model never chooses a tool and never
|
|
14
|
+
sees anything but the finished contract.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
|
|
21
|
+
from strikeone.ai import evidence as ev_mod
|
|
22
|
+
from strikeone.ai import validator as val_mod
|
|
23
|
+
from strikeone.ai.providers import AIProvider
|
|
24
|
+
|
|
25
|
+
SYSTEM_PROMPT = """\
|
|
26
|
+
You are the narration layer of a deterministic fraud decision engine.
|
|
27
|
+
Every decision was already made by the engine; you never compute, judge,
|
|
28
|
+
recommend, or second-guess. You explain, citing evidence.
|
|
29
|
+
|
|
30
|
+
Output format (a validator drops anything else, so follow it exactly):
|
|
31
|
+
- 3 to 8 lines total.
|
|
32
|
+
- Every line that states a fact MUST be exactly:
|
|
33
|
+
CLAIM: <evidence id> | <the value exactly as written in the evidence> | <one plain sentence that uses that value naturally>
|
|
34
|
+
- You may add at most 2 lines of the form:
|
|
35
|
+
SUMMARY: <a sentence with NO digits at all>
|
|
36
|
+
- Use only ids that appear in the evidence list. Do not invent numbers.
|
|
37
|
+
- Do not mention these rules, the ids' letter-number form, or the word
|
|
38
|
+
"evidence" inside the sentences themselves.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
TASK_HINTS = {
|
|
42
|
+
"why": ("Explain why this transaction got this decision. Cover the "
|
|
43
|
+
"decision and lane, the entity's history, and how the amount "
|
|
44
|
+
"and probability compare to their baselines."),
|
|
45
|
+
"timeline": ("Narrate this case in order: the quiet period, the first "
|
|
46
|
+
"labelled transaction, and the run after it that a "
|
|
47
|
+
"standing blocklist would also have covered."),
|
|
48
|
+
"compare": ("Explain what each of the two systems did with this "
|
|
49
|
+
"transaction and why they agreed or diverged: the "
|
|
50
|
+
"blocklist state, where the score ranks against the "
|
|
51
|
+
"review-budget cutoff, and each verdict."),
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def user_prompt(contract: dict) -> str:
|
|
56
|
+
return (f"Task: {TASK_HINTS[contract['command']]}\n\n"
|
|
57
|
+
"The evidence contract (your ONLY source of facts):\n"
|
|
58
|
+
+ json.dumps(contract, indent=2, ensure_ascii=False))
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def run(command: str, df, mapping, target, provider: AIProvider,
|
|
62
|
+
capacity_per_day: int = 100) -> dict:
|
|
63
|
+
"""Returns {contract, raw, validated, rendered, model, provider}."""
|
|
64
|
+
builder = ev_mod.BUILDERS[command] # deterministic router
|
|
65
|
+
if command == "compare":
|
|
66
|
+
contract = builder(df, mapping, target,
|
|
67
|
+
capacity_per_day=capacity_per_day)
|
|
68
|
+
else:
|
|
69
|
+
contract = builder(df, mapping, target)
|
|
70
|
+
reply = provider.narrate(SYSTEM_PROMPT, user_prompt(contract))
|
|
71
|
+
v = val_mod.validate(reply.text, contract)
|
|
72
|
+
rendered = val_mod.render(v, contract, reply.model, reply.provider_label)
|
|
73
|
+
return {"contract": contract, "raw": reply.text, "validated": v,
|
|
74
|
+
"rendered": rendered, "model": reply.model,
|
|
75
|
+
"provider": reply.provider_label,
|
|
76
|
+
"validity": v.validity,
|
|
77
|
+
"evidence_hash": contract["evidence_hash"]}
|
strikeone/ai/evidence.py
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""The evidence contract — item 1 of the AI layer, built first, frozen.
|
|
2
|
+
|
|
3
|
+
A deterministic JSON document produced entirely by the existing engine
|
|
4
|
+
BEFORE any model is asked to speak. Every provider receives exactly this
|
|
5
|
+
and nothing else. The model never adds, removes or alters a field.
|
|
6
|
+
|
|
7
|
+
Frozen schema, contract_version 1.0 (top-level keys, exactly these):
|
|
8
|
+
|
|
9
|
+
contract_version, evidence_hash, command, transaction_id, case_id,
|
|
10
|
+
decision, lane, fraud_probability, episode_state, evidence, policy
|
|
11
|
+
|
|
12
|
+
Each evidence item has exactly: id, feature, value, baseline, source.
|
|
13
|
+
`evidence_hash` is the sha256 of the canonicalised contract (sorted keys,
|
|
14
|
+
compact separators, the hash field itself excluded), so any narration can
|
|
15
|
+
be traced to the exact evidence that produced it.
|
|
16
|
+
|
|
17
|
+
Invariants (asserted by tests):
|
|
18
|
+
- no raw transaction rows: evidence carries named, derived facts only;
|
|
19
|
+
- the sealed holdout is never read (strikeone.seal is not touched here);
|
|
20
|
+
- built twice on the same frame, the contract is byte-identical.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import hashlib
|
|
26
|
+
import json
|
|
27
|
+
|
|
28
|
+
import numpy as np
|
|
29
|
+
import pandas as pd
|
|
30
|
+
|
|
31
|
+
from strikeone import entity as entity_mod
|
|
32
|
+
from strikeone import metrics as M
|
|
33
|
+
from strikeone.contract import ContractError, Mapping
|
|
34
|
+
from strikeone.policy_engine import CENTRAL
|
|
35
|
+
|
|
36
|
+
CONTRACT_VERSION = "1.0"
|
|
37
|
+
TOP_KEYS = ["contract_version", "evidence_hash", "command",
|
|
38
|
+
"transaction_id", "case_id", "decision", "lane",
|
|
39
|
+
"fraud_probability", "episode_state", "evidence", "policy"]
|
|
40
|
+
ITEM_KEYS = ["id", "feature", "value", "baseline", "source"]
|
|
41
|
+
ACTION_NAMES = {0: "APPROVE", 1: "STEP_UP", 2: "BLOCK"}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _round(v):
|
|
45
|
+
if isinstance(v, (float, np.floating)):
|
|
46
|
+
return round(float(v), 4)
|
|
47
|
+
if isinstance(v, (int, np.integer)):
|
|
48
|
+
return int(v)
|
|
49
|
+
return v
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def canonical_hash(contract: dict) -> str:
|
|
53
|
+
body = {k: v for k, v in contract.items() if k != "evidence_hash"}
|
|
54
|
+
blob = json.dumps(body, sort_keys=True, separators=(",", ":"),
|
|
55
|
+
ensure_ascii=False)
|
|
56
|
+
return hashlib.sha256(blob.encode("utf-8")).hexdigest()
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _finish(contract: dict) -> dict:
|
|
60
|
+
assert list(contract) == TOP_KEYS or set(contract) == set(TOP_KEYS)
|
|
61
|
+
for item in contract["evidence"]:
|
|
62
|
+
assert list(item) == ITEM_KEYS, f"evidence item keys drifted: {item}"
|
|
63
|
+
contract["evidence_hash"] = canonical_hash(contract)
|
|
64
|
+
return {k: contract[k] for k in TOP_KEYS}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _prepare(df: pd.DataFrame, mapping: Mapping):
|
|
68
|
+
"""Engine-side derived state shared by all three builders."""
|
|
69
|
+
d = df.sort_values(["t", "transaction_id"]).reset_index(drop=True)
|
|
70
|
+
has_label = "label" in d.columns and not d["label"].isna().any()
|
|
71
|
+
if not has_label:
|
|
72
|
+
raise ContractError(
|
|
73
|
+
"the AI layer explains decisions against labelled history; "
|
|
74
|
+
"map a label column first (--map label=<col>)")
|
|
75
|
+
y = d["label"].to_numpy().astype(int)
|
|
76
|
+
ent = d["entity"].astype(str).to_numpy()
|
|
77
|
+
t = d["t"].to_numpy().astype(np.int64)
|
|
78
|
+
tb = d["transaction_id"].to_numpy()
|
|
79
|
+
delay = float(mapping.label_delay_days)
|
|
80
|
+
bl = entity_mod.pit_delayed_label_stats(
|
|
81
|
+
pd.Series(ent), t, y, tb, delay_days=delay, prefix="u")
|
|
82
|
+
knowable = np.nan_to_num(bl["u_fraud_rate"].to_numpy()
|
|
83
|
+
* bl["u_labeled_cnt"].to_numpy())
|
|
84
|
+
flag = knowable > 0
|
|
85
|
+
grp = pd.Series(y).groupby(pd.Series(ent))
|
|
86
|
+
prior_frauds_any_age = (grp.cumsum() - pd.Series(y)).to_numpy()
|
|
87
|
+
prior_txns = grp.cumcount().to_numpy()
|
|
88
|
+
return d, y, ent, t, delay, flag, knowable, prior_frauds_any_age, prior_txns
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _policy_block(d: pd.DataFrame) -> dict | None:
|
|
92
|
+
if "p" not in d.columns or d["p"].isna().all():
|
|
93
|
+
return None
|
|
94
|
+
prm = M.CostParams(**CENTRAL)
|
|
95
|
+
ec = M.expected_cost_matrix(d["p"].to_numpy(float),
|
|
96
|
+
d["amount"].to_numpy(float), prm)
|
|
97
|
+
mix = np.bincount(ec.argmin(axis=1), minlength=3)
|
|
98
|
+
return {"approve": int(mix[0]), "step_up": int(mix[1]),
|
|
99
|
+
"block": int(mix[2]),
|
|
100
|
+
"params": {k: _round(v) for k, v in CENTRAL.items()}}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def build_why(df: pd.DataFrame, mapping: Mapping, transaction_id) -> dict:
|
|
104
|
+
d, y, ent, t, delay, flag, knowable, prior_any, prior_n = \
|
|
105
|
+
_prepare(df, mapping)
|
|
106
|
+
hits = np.where(d["transaction_id"].astype(str).to_numpy()
|
|
107
|
+
== str(transaction_id))[0]
|
|
108
|
+
if len(hits) == 0:
|
|
109
|
+
raise ContractError(f"transaction {transaction_id!r} not found")
|
|
110
|
+
i = int(hits[0])
|
|
111
|
+
amount = float(d["amount"].iloc[i])
|
|
112
|
+
e_rows = np.where(ent == ent[i])[0]
|
|
113
|
+
prior_rows = e_rows[e_rows < i]
|
|
114
|
+
prior_amt_mean = (float(d["amount"].iloc[prior_rows].mean())
|
|
115
|
+
if len(prior_rows) else None)
|
|
116
|
+
|
|
117
|
+
lane = 1 if flag[i] else 2
|
|
118
|
+
p_i = None
|
|
119
|
+
if "p" in d.columns and not pd.isna(d["p"].iloc[i]):
|
|
120
|
+
p_i = float(d["p"].iloc[i])
|
|
121
|
+
if lane == 1:
|
|
122
|
+
decision = "BLOCK"
|
|
123
|
+
elif p_i is not None:
|
|
124
|
+
prm = M.CostParams(**CENTRAL)
|
|
125
|
+
ec = M.expected_cost_matrix(np.array([p_i]), np.array([amount]), prm)
|
|
126
|
+
decision = ACTION_NAMES[int(ec.argmin(axis=1)[0])]
|
|
127
|
+
else:
|
|
128
|
+
decision = None
|
|
129
|
+
if flag[i]:
|
|
130
|
+
state = "already flagged"
|
|
131
|
+
elif y[i] == 1 and prior_any[i] == 0:
|
|
132
|
+
state = "first attempt"
|
|
133
|
+
else:
|
|
134
|
+
state = "no prior flags"
|
|
135
|
+
|
|
136
|
+
ev = [
|
|
137
|
+
{"id": "F1", "feature": "decision", "value": decision,
|
|
138
|
+
"baseline": None,
|
|
139
|
+
"source": "expected-cost argmin at frozen central params"
|
|
140
|
+
if lane == 2 else "lane-1 blocklist rule"},
|
|
141
|
+
{"id": "F2", "feature": "lane", "value": lane, "baseline": None,
|
|
142
|
+
"source": "two-lane router (point-in-time blocklist)"},
|
|
143
|
+
{"id": "F3", "feature": "episode_state", "value": state,
|
|
144
|
+
"baseline": None, "source": "episode roles, global stream"},
|
|
145
|
+
{"id": "F4", "feature": "prior_transactions_on_entity",
|
|
146
|
+
"value": int(prior_n[i]),
|
|
147
|
+
"baseline": _round(float(np.median(prior_n))),
|
|
148
|
+
"source": "point-in-time entity history (strikeone.entity)"},
|
|
149
|
+
{"id": "F5", "feature": "knowable_prior_frauds_on_entity",
|
|
150
|
+
"value": int(round(knowable[i])),
|
|
151
|
+
"baseline": _round(float(knowable.mean())),
|
|
152
|
+
"source": f"labels at least {delay:g} days old at decision time"},
|
|
153
|
+
{"id": "F6", "feature": "amount", "value": _round(amount),
|
|
154
|
+
"baseline": _round(float(d["amount"].median())),
|
|
155
|
+
"source": "amount column vs population median"},
|
|
156
|
+
]
|
|
157
|
+
if prior_amt_mean is not None:
|
|
158
|
+
ev.append({"id": "F7", "feature": "entity_prior_mean_amount",
|
|
159
|
+
"value": _round(prior_amt_mean), "baseline": None,
|
|
160
|
+
"source": "mean amount of this entity's earlier rows"})
|
|
161
|
+
if p_i is not None:
|
|
162
|
+
ev.append({"id": "F8", "feature": "fraud_probability",
|
|
163
|
+
"value": _round(p_i),
|
|
164
|
+
"baseline": _round(float(d["p"].mean())),
|
|
165
|
+
"source": "calibrated p column (baseline: its mean)"})
|
|
166
|
+
if "score" in d.columns and not pd.isna(d["score"].iloc[i]):
|
|
167
|
+
pct = float((d["score"] < d["score"].iloc[i]).mean() * 100)
|
|
168
|
+
ev.append({"id": "F9", "feature": "score_percentile",
|
|
169
|
+
"value": _round(pct), "baseline": 50.0,
|
|
170
|
+
"source": "rank of the score column within this file"})
|
|
171
|
+
|
|
172
|
+
return _finish({
|
|
173
|
+
"contract_version": CONTRACT_VERSION, "evidence_hash": "",
|
|
174
|
+
"command": "why", "transaction_id": str(transaction_id),
|
|
175
|
+
"case_id": None, "decision": decision, "lane": lane,
|
|
176
|
+
"fraud_probability": _round(p_i) if p_i is not None else None,
|
|
177
|
+
"episode_state": state, "evidence": ev, "policy": _policy_block(d),
|
|
178
|
+
})
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def build_timeline(df: pd.DataFrame, mapping: Mapping, case_id) -> dict:
|
|
182
|
+
d, y, ent, t, delay, flag, knowable, prior_any, prior_n = \
|
|
183
|
+
_prepare(df, mapping)
|
|
184
|
+
rows = np.where(ent == str(case_id))[0]
|
|
185
|
+
if len(rows) == 0:
|
|
186
|
+
raise ContractError(f"case (entity) {case_id!r} not found")
|
|
187
|
+
yy = y[rows]
|
|
188
|
+
n_fraud = int(yy.sum())
|
|
189
|
+
day = (t[rows] - t.min()) / 86400.0
|
|
190
|
+
ev = [
|
|
191
|
+
{"id": "T1", "feature": "case_transactions", "value": int(len(rows)),
|
|
192
|
+
"baseline": None, "source": "all rows for this entity"},
|
|
193
|
+
]
|
|
194
|
+
if n_fraud == 0:
|
|
195
|
+
state = "no prior flags"
|
|
196
|
+
ev.append({"id": "T2", "feature": "labelled_frauds_in_case",
|
|
197
|
+
"value": 0, "baseline": None,
|
|
198
|
+
"source": "label column over the case"})
|
|
199
|
+
else:
|
|
200
|
+
first = int(np.argmax(yy == 1))
|
|
201
|
+
quiet = int(first)
|
|
202
|
+
coverable = int(flag[rows][yy == 1].sum())
|
|
203
|
+
state = "already flagged" if n_fraud > 1 else "first attempt"
|
|
204
|
+
ev += [
|
|
205
|
+
{"id": "T2", "feature": "quiet_transactions_before_first_fraud",
|
|
206
|
+
"value": quiet, "baseline": None,
|
|
207
|
+
"source": "rows before the case's first labelled transaction"},
|
|
208
|
+
{"id": "T3", "feature": "first_fraud_day_index",
|
|
209
|
+
"value": _round(float(day[first])), "baseline": None,
|
|
210
|
+
"source": "days since the start of this file"},
|
|
211
|
+
{"id": "T4", "feature": "first_fraud_amount",
|
|
212
|
+
"value": _round(float(d["amount"].iloc[rows[first]])),
|
|
213
|
+
"baseline": _round(float(d["amount"].iloc[rows[:first]].mean()))
|
|
214
|
+
if first else None,
|
|
215
|
+
"source": "amount vs the case's own quiet-period mean"},
|
|
216
|
+
{"id": "T5", "feature": "labelled_frauds_in_case",
|
|
217
|
+
"value": n_fraud, "baseline": None,
|
|
218
|
+
"source": "label column over the case"},
|
|
219
|
+
{"id": "T6", "feature": "blocklist_coverable_in_case",
|
|
220
|
+
"value": coverable, "baseline": None,
|
|
221
|
+
"source": f"fraud rows where a {delay:g}-day-delayed blocklist "
|
|
222
|
+
"already knew this entity"},
|
|
223
|
+
{"id": "T7", "feature": "case_fraud_amount_total",
|
|
224
|
+
"value": _round(float(d["amount"].iloc[rows][yy == 1].sum())),
|
|
225
|
+
"baseline": None, "source": "sum of labelled-fraud amounts"},
|
|
226
|
+
]
|
|
227
|
+
return _finish({
|
|
228
|
+
"contract_version": CONTRACT_VERSION, "evidence_hash": "",
|
|
229
|
+
"command": "timeline", "transaction_id": None,
|
|
230
|
+
"case_id": str(case_id), "decision": None, "lane": None,
|
|
231
|
+
"fraud_probability": None, "episode_state": state,
|
|
232
|
+
"evidence": ev, "policy": None,
|
|
233
|
+
})
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def build_compare(df: pd.DataFrame, mapping: Mapping, transaction_id,
|
|
237
|
+
capacity_per_day: int = 100) -> dict:
|
|
238
|
+
d, y, ent, t, delay, flag, knowable, prior_any, prior_n = \
|
|
239
|
+
_prepare(df, mapping)
|
|
240
|
+
if "score" not in d.columns or d["score"].isna().all():
|
|
241
|
+
raise ContractError("compare needs a score column (--map score=...)")
|
|
242
|
+
hits = np.where(d["transaction_id"].astype(str).to_numpy()
|
|
243
|
+
== str(transaction_id))[0]
|
|
244
|
+
if len(hits) == 0:
|
|
245
|
+
raise ContractError(f"transaction {transaction_id!r} not found")
|
|
246
|
+
i = int(hits[0])
|
|
247
|
+
days = max((t.max() - t.min()) / 86400.0, 1.0)
|
|
248
|
+
budget = int(round(capacity_per_day * days))
|
|
249
|
+
s = d["score"].to_numpy(float)
|
|
250
|
+
single_alert = M.alerts_at_budget(s, budget)
|
|
251
|
+
lane2 = ~flag
|
|
252
|
+
k2 = min(budget, int(lane2.sum()))
|
|
253
|
+
two_alert = flag | M.alerts_at_budget(np.where(lane2, s, -np.inf), k2)
|
|
254
|
+
cutoff = float(np.sort(s)[-budget]) if budget <= len(s) else float("-inf")
|
|
255
|
+
single_v = "alert" if single_alert[i] else "no alert"
|
|
256
|
+
two_v = ("blocked by the lane-1 blocklist" if flag[i]
|
|
257
|
+
else ("alert" if two_alert[i] else "no alert"))
|
|
258
|
+
state = ("already flagged" if flag[i]
|
|
259
|
+
else ("first attempt" if y[i] == 1 and prior_any[i] == 0
|
|
260
|
+
else "no prior flags"))
|
|
261
|
+
ev = [
|
|
262
|
+
{"id": "C1", "feature": "blocklist_state",
|
|
263
|
+
"value": "flagged" if flag[i] else "not flagged", "baseline": None,
|
|
264
|
+
"source": f"point-in-time blocklist, {delay:g}-day label delay"},
|
|
265
|
+
{"id": "C2", "feature": "score", "value": _round(float(s[i])),
|
|
266
|
+
"baseline": _round(float(s.mean())),
|
|
267
|
+
"source": "score column (baseline: its mean)"},
|
|
268
|
+
{"id": "C3", "feature": "score_percentile",
|
|
269
|
+
"value": _round(float((s < s[i]).mean() * 100)), "baseline": 50.0,
|
|
270
|
+
"source": "rank of the score within this file"},
|
|
271
|
+
{"id": "C4", "feature": "review_budget_cutoff_score",
|
|
272
|
+
"value": _round(cutoff), "baseline": None,
|
|
273
|
+
"source": f"top-{budget:,} alerts at {capacity_per_day}/day "
|
|
274
|
+
f"over {days:.1f} days"},
|
|
275
|
+
{"id": "C5", "feature": "single_lane_scorer_verdict",
|
|
276
|
+
"value": single_v, "baseline": None,
|
|
277
|
+
"source": "score ranking alone, same budget"},
|
|
278
|
+
{"id": "C6", "feature": "two_lane_system_verdict",
|
|
279
|
+
"value": two_v, "baseline": None,
|
|
280
|
+
"source": "blocklist lane first, scorer on the rest"},
|
|
281
|
+
]
|
|
282
|
+
# the divergence mechanism is determined by the ENGINE, not left to
|
|
283
|
+
# the model's interpretation (a digit-free wrong reading slips past a
|
|
284
|
+
# numeric validator; a citable string does not)
|
|
285
|
+
if single_alert[i] == two_alert[i] and not flag[i]:
|
|
286
|
+
mech = "both systems reached the same verdict; no divergence"
|
|
287
|
+
elif flag[i]:
|
|
288
|
+
mech = ("the blocklist lane knew this entity from a prior "
|
|
289
|
+
"labelled fraud; the score ranking is irrelevant in "
|
|
290
|
+
"lane 1")
|
|
291
|
+
elif two_alert[i] and not single_alert[i]:
|
|
292
|
+
mech = ("routing freed capacity: the same review budget spread "
|
|
293
|
+
"over fewer lane-2 candidates lowers the cutoff below "
|
|
294
|
+
"this score")
|
|
295
|
+
else:
|
|
296
|
+
mech = ("the single-lane ranking spent budget on rows the "
|
|
297
|
+
"blocklist lane would have absorbed, reaching deeper "
|
|
298
|
+
"into the file")
|
|
299
|
+
ev.append({"id": "C7", "feature": "divergence_mechanism",
|
|
300
|
+
"value": mech, "baseline": None,
|
|
301
|
+
"source": "deterministic comparison of the two alert sets"})
|
|
302
|
+
return _finish({
|
|
303
|
+
"contract_version": CONTRACT_VERSION, "evidence_hash": "",
|
|
304
|
+
"command": "compare", "transaction_id": str(transaction_id),
|
|
305
|
+
"case_id": None, "decision": two_v, "lane": 1 if flag[i] else 2,
|
|
306
|
+
"fraud_probability": None, "episode_state": state,
|
|
307
|
+
"evidence": ev, "policy": None,
|
|
308
|
+
})
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
# The deterministic tool router (item 3): CLI intent -> builder. The model
|
|
312
|
+
# is not in this dict and never sees it; a hallucinated tool choice is
|
|
313
|
+
# impossible by construction.
|
|
314
|
+
BUILDERS = {"why": build_why, "timeline": build_timeline,
|
|
315
|
+
"compare": build_compare}
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""One interface, two adapters — item 3.
|
|
2
|
+
|
|
3
|
+
AIProvider (abstract)
|
|
4
|
+
├── OllamaProvider local, http://localhost:11434
|
|
5
|
+
└── OpenAICompatibleProvider base_url + model slug
|
|
6
|
+
|
|
7
|
+
The second adapter covers OpenAI, OpenRouter, Ollama Cloud and any custom
|
|
8
|
+
endpoint; there is deliberately no bespoke per-vendor adapter. Credentials
|
|
9
|
+
are env vars only (item 5): the config stores the NAME of the env var,
|
|
10
|
+
never a value. Every reply records which model actually answered — an
|
|
11
|
+
explanation that doesn't name its author is a hole.
|
|
12
|
+
|
|
13
|
+
No third-party HTTP dependency: urllib from the standard library.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
import urllib.error
|
|
21
|
+
import urllib.request
|
|
22
|
+
from abc import ABC, abstractmethod
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ProviderError(RuntimeError):
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Reply:
|
|
32
|
+
text: str
|
|
33
|
+
model: str # the model that actually answered
|
|
34
|
+
provider_label: str
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _post_json(url: str, payload: dict, headers: dict, timeout: float) -> dict:
|
|
38
|
+
req = urllib.request.Request(
|
|
39
|
+
url, data=json.dumps(payload).encode("utf-8"),
|
|
40
|
+
headers={"Content-Type": "application/json", **headers},
|
|
41
|
+
method="POST")
|
|
42
|
+
try:
|
|
43
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
44
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
45
|
+
except urllib.error.URLError as e:
|
|
46
|
+
raise ProviderError(f"provider unreachable at {url}: {e}") from e
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AIProvider(ABC):
|
|
50
|
+
"""Narrates a finished evidence contract. Never chooses tools, never
|
|
51
|
+
computes: the deterministic router ran before this object was called."""
|
|
52
|
+
|
|
53
|
+
@abstractmethod
|
|
54
|
+
def narrate(self, system_prompt: str, user_prompt: str) -> Reply: ...
|
|
55
|
+
|
|
56
|
+
@abstractmethod
|
|
57
|
+
def chain_text(self) -> str:
|
|
58
|
+
"""Item 6: show the evidence path, not just the destination."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class OllamaProvider(AIProvider):
|
|
62
|
+
def __init__(self, model: str, base_url: str = "http://localhost:11434",
|
|
63
|
+
timeout: float = 180.0, think: bool | None = None):
|
|
64
|
+
self.model = model
|
|
65
|
+
self.base_url = base_url.rstrip("/")
|
|
66
|
+
self.timeout = timeout
|
|
67
|
+
self.think = think # False = ask hybrid-reasoning models to skip
|
|
68
|
+
# the thinking pass (narration needs none)
|
|
69
|
+
|
|
70
|
+
def narrate(self, system_prompt: str, user_prompt: str) -> Reply:
|
|
71
|
+
payload = {"model": self.model, "stream": False,
|
|
72
|
+
"options": {"temperature": 0.0},
|
|
73
|
+
"messages": [{"role": "system", "content": system_prompt},
|
|
74
|
+
{"role": "user", "content": user_prompt}]}
|
|
75
|
+
if self.think is not None:
|
|
76
|
+
payload["think"] = self.think
|
|
77
|
+
try:
|
|
78
|
+
out = _post_json(f"{self.base_url}/api/chat", payload,
|
|
79
|
+
{}, self.timeout)
|
|
80
|
+
except ProviderError:
|
|
81
|
+
if "think" not in payload:
|
|
82
|
+
raise
|
|
83
|
+
payload.pop("think") # model may not support the flag
|
|
84
|
+
out = _post_json(f"{self.base_url}/api/chat", payload,
|
|
85
|
+
{}, self.timeout)
|
|
86
|
+
return Reply(text=out.get("message", {}).get("content", ""),
|
|
87
|
+
model=out.get("model", self.model),
|
|
88
|
+
provider_label="ollama, local")
|
|
89
|
+
|
|
90
|
+
def chain_text(self) -> str:
|
|
91
|
+
return "\n".join([
|
|
92
|
+
f"Provider: Ollama (local) Model: {self.model}",
|
|
93
|
+
f"Endpoint: {self.base_url}",
|
|
94
|
+
"Evidence leaves this machine: NO",
|
|
95
|
+
])
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class OpenAICompatibleProvider(AIProvider):
|
|
99
|
+
"""OpenAI, OpenRouter, Ollama Cloud, or any /v1-compatible endpoint."""
|
|
100
|
+
|
|
101
|
+
def __init__(self, model: str, base_url: str,
|
|
102
|
+
api_key_env: str = "OPENAI_API_KEY", timeout: float = 120.0):
|
|
103
|
+
self.model = model
|
|
104
|
+
self.base_url = base_url.rstrip("/")
|
|
105
|
+
self.api_key_env = api_key_env # the NAME; the value stays in env
|
|
106
|
+
self.timeout = timeout
|
|
107
|
+
|
|
108
|
+
def _key(self) -> str:
|
|
109
|
+
key = os.environ.get(self.api_key_env, "")
|
|
110
|
+
if not key:
|
|
111
|
+
raise ProviderError(
|
|
112
|
+
f"env var {self.api_key_env} is not set. Export it and "
|
|
113
|
+
"retry; strikeone never stores or prompts for secrets.")
|
|
114
|
+
return key
|
|
115
|
+
|
|
116
|
+
def narrate(self, system_prompt: str, user_prompt: str) -> Reply:
|
|
117
|
+
out = _post_json(
|
|
118
|
+
f"{self.base_url}/chat/completions",
|
|
119
|
+
{"model": self.model, "temperature": 0.0,
|
|
120
|
+
"messages": [{"role": "system", "content": system_prompt},
|
|
121
|
+
{"role": "user", "content": user_prompt}]},
|
|
122
|
+
{"Authorization": f"Bearer {self._key()}"}, self.timeout)
|
|
123
|
+
try:
|
|
124
|
+
text = out["choices"][0]["message"]["content"]
|
|
125
|
+
except (KeyError, IndexError) as e:
|
|
126
|
+
raise ProviderError(f"unexpected response shape: {out}") from e
|
|
127
|
+
answered = out.get("model", self.model) # aggregators may rewrite
|
|
128
|
+
return Reply(text=text, model=answered,
|
|
129
|
+
provider_label=f"{self._host()}, remote")
|
|
130
|
+
|
|
131
|
+
def _host(self) -> str:
|
|
132
|
+
return self.base_url.split("//", 1)[-1].split("/", 1)[0]
|
|
133
|
+
|
|
134
|
+
def chain_text(self) -> str:
|
|
135
|
+
host = self._host()
|
|
136
|
+
lines = [f"Provider: {host} Model: {self.model}"]
|
|
137
|
+
if "openrouter" in host:
|
|
138
|
+
upstream = self.model.split("/", 1)[0] if "/" in self.model \
|
|
139
|
+
else "the routed provider"
|
|
140
|
+
lines += [
|
|
141
|
+
f"Evidence path: this machine → {host} → {upstream}",
|
|
142
|
+
"(an aggregator is two parties, not one)",
|
|
143
|
+
]
|
|
144
|
+
else:
|
|
145
|
+
lines.append(f"Evidence path: this machine → {host}")
|
|
146
|
+
lines += [
|
|
147
|
+
"Sent: decision evidence only (no raw transactions, no "
|
|
148
|
+
"holdout data)",
|
|
149
|
+
f"Credential: env var {self.api_key_env} "
|
|
150
|
+
"(never stored, never prompted for)",
|
|
151
|
+
]
|
|
152
|
+
return "\n".join(lines)
|