robustrep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- robustrep/__init__.py +6 -0
- robustrep/aggregate.py +262 -0
- robustrep/cli.py +469 -0
- robustrep/config.py +58 -0
- robustrep/evidence.py +69 -0
- robustrep/explain.py +112 -0
- robustrep/normalize.py +73 -0
- robustrep/pipeline.py +195 -0
- robustrep/report/__init__.py +0 -0
- robustrep/report/adversarial.py +245 -0
- robustrep/report/export.py +52 -0
- robustrep/report/figures.py +140 -0
- robustrep/report/render.py +249 -0
- robustrep/report/sensitivity.py +188 -0
- robustrep/schema.py +109 -0
- robustrep/sources/__init__.py +0 -0
- robustrep/sources/base_erc8004.py +390 -0
- robustrep/sources/evidence_fetch.py +395 -0
- robustrep/sources/rater_profile.py +272 -0
- robustrep/sources/rpc.py +286 -0
- robustrep/store.py +305 -0
- robustrep/sybil.py +260 -0
- robustrep-0.1.0.dist-info/METADATA +34 -0
- robustrep-0.1.0.dist-info/RECORD +28 -0
- robustrep-0.1.0.dist-info/WHEEL +5 -0
- robustrep-0.1.0.dist-info/entry_points.txt +2 -0
- robustrep-0.1.0.dist-info/licenses/LICENSE +202 -0
- robustrep-0.1.0.dist-info/top_level.txt +1 -0
robustrep/cli.py
ADDED
|
@@ -0,0 +1,469 @@
|
|
|
1
|
+
"""``robustrep fetch`` / ``robustrep score`` command-line entry points.
|
|
2
|
+
|
|
3
|
+
Kept intentionally thin: this module only orchestrates calls into the
|
|
4
|
+
library (``robustrep.sources.*``, ``robustrep.pipeline``, ``robustrep.sybil``,
|
|
5
|
+
``robustrep.store``) and never re-implements their business logic. ``fetch``'s
|
|
6
|
+
pipeline is broken into small private ``_step_*`` helpers (one per stage:
|
|
7
|
+
sync, block timestamps, agent owners, rater profiles, evidence) that each
|
|
8
|
+
return a one-line summary string -- this lets tests monkeypatch the
|
|
9
|
+
``base``/``enrich_raters``/``classify_all`` symbols in this module and drive
|
|
10
|
+
the whole command end-to-end without any network access.
|
|
11
|
+
|
|
12
|
+
``report`` generates figures 1-5, ``report.md`` and ``scores.json`` under
|
|
13
|
+
``out_dir/<block>/``, then publishes an atomic copy to ``out_dir/latest/``.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import logging
|
|
18
|
+
import os
|
|
19
|
+
import shutil
|
|
20
|
+
import sys
|
|
21
|
+
import tempfile
|
|
22
|
+
from contextlib import contextmanager
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import List, Optional
|
|
25
|
+
|
|
26
|
+
import numpy
|
|
27
|
+
import pandas
|
|
28
|
+
import typer
|
|
29
|
+
|
|
30
|
+
from . import __version__
|
|
31
|
+
from .config import Config
|
|
32
|
+
from .pipeline import score as score_fn
|
|
33
|
+
from .report.adversarial import scenario_table
|
|
34
|
+
from .report.export import export_json
|
|
35
|
+
from .report.figures import fig_evidence, fig_mean_vs_robust, fig_rank_shift, fig_sybil_clusters
|
|
36
|
+
from .report.render import evidence_level_shares, render_markdown
|
|
37
|
+
from .report.sensitivity import fig_sensitivity, sensitivity_table
|
|
38
|
+
from .sources import base_erc8004 as base
|
|
39
|
+
from .sources.evidence_fetch import classify_all
|
|
40
|
+
from .sources.rater_profile import DEFAULT_RPS, EtherscanClient, enrich_raters, estimate_seconds
|
|
41
|
+
from .sources.rpc import RpcClient, RpcError
|
|
42
|
+
from .store import Store
|
|
43
|
+
from .sybil import cluster_raters, profiles_from_records
|
|
44
|
+
|
|
45
|
+
app = typer.Typer(add_completion=False, help="Robust, transport-agnostic reputation scoring for AI agents.")
|
|
46
|
+
|
|
47
|
+
logger = logging.getLogger(__name__)
|
|
48
|
+
|
|
49
|
+
# Agent-owner resolution progress is logged (INFO) every this many agents.
|
|
50
|
+
OWNER_LOG_EVERY = 500
|
|
51
|
+
|
|
52
|
+
# Default number of agents resolved per JSON-RPC batch in _step_owners (see
|
|
53
|
+
# --owner-batch).
|
|
54
|
+
OWNER_BATCH_DEFAULT = 100
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@app.callback()
|
|
58
|
+
def main(verbose: bool = typer.Option(False, "--verbose", help="Enable DEBUG-level logging.")) -> None:
|
|
59
|
+
"""Robust, transport-agnostic reputation scoring for AI agents."""
|
|
60
|
+
# Only ever *adjust the level* of our own package logger -- never call
|
|
61
|
+
# logging.basicConfig(..., force=True), which tears down and replaces every
|
|
62
|
+
# handler on the root logger (including one belonging to a caller embedding
|
|
63
|
+
# this CLI, or pytest's caplog) with a fresh StreamHandler bound to
|
|
64
|
+
# whatever sys.stderr happens to be *at this exact call*. Under repeated
|
|
65
|
+
# `typer.testing.CliRunner.invoke()` calls in one process, that snapshot is
|
|
66
|
+
# a redirected stream that gets closed once the invocation ends, so a
|
|
67
|
+
# later log call reusing that handler raises "I/O operation on closed
|
|
68
|
+
# file." Adding a handler only when the root logger has none yet avoids
|
|
69
|
+
# ever rebinding a handler to a stream that may later be closed.
|
|
70
|
+
#
|
|
71
|
+
# --verbose sets DEBUG (not INFO): _rpc_guard's "re-run with --verbose for
|
|
72
|
+
# detail" message promises that --verbose actually surfaces the
|
|
73
|
+
# DEBUG-level detail rpc.py and _rpc_guard log on failure; INFO wouldn't.
|
|
74
|
+
#
|
|
75
|
+
# Non-verbose resets to NOTSET (defer to the root logger's own level,
|
|
76
|
+
# WARNING by default) rather than pinning WARNING explicitly on this
|
|
77
|
+
# named logger: an explicit level here would otherwise outlive this call
|
|
78
|
+
# (loggers are process-global singletons) and, unlike NOTSET, would block
|
|
79
|
+
# a later `caplog.at_level(logging.INFO)` (which only raises the ROOT
|
|
80
|
+
# logger's level) from ever reaching child loggers under "robustrep" --
|
|
81
|
+
# Python's level lookup stops at the first ancestor with an explicit
|
|
82
|
+
# level, so an explicit WARNING here would shadow root's override.
|
|
83
|
+
logging.getLogger("robustrep").setLevel(logging.DEBUG if verbose else logging.NOTSET)
|
|
84
|
+
root = logging.getLogger()
|
|
85
|
+
if not root.handlers:
|
|
86
|
+
root.addHandler(logging.StreamHandler(sys.stderr))
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _no_blank(value: Optional[str]) -> Optional[str]:
|
|
90
|
+
"""Typer/click callback: reject an explicitly-empty/whitespace-only option
|
|
91
|
+
value (e.g. ``--etherscan-key ""``) as a usage error rather than silently
|
|
92
|
+
treating it as "not given"."""
|
|
93
|
+
if value is not None and value.strip() == "":
|
|
94
|
+
raise typer.BadParameter("must not be empty")
|
|
95
|
+
return value
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@contextmanager
|
|
99
|
+
def _rpc_guard():
|
|
100
|
+
"""Convert an ``RpcError`` escaping an RPC-touching step into a clean CLI
|
|
101
|
+
failure: exit 3, echoing only the already-redacted message (see
|
|
102
|
+
``robustrep.sources.rpc._redact`` -- never the raw exception, which is
|
|
103
|
+
what ``str(e)`` on an ``RpcError`` already guarantees) plus a pointer to
|
|
104
|
+
``--verbose``. The full exception (traceback included) still reaches the
|
|
105
|
+
DEBUG log, which ``--verbose`` turns on.
|
|
106
|
+
"""
|
|
107
|
+
try:
|
|
108
|
+
yield
|
|
109
|
+
except RpcError as e:
|
|
110
|
+
typer.echo(f"ERROR: RPC failed: {e} (re-run with --verbose for detail)")
|
|
111
|
+
logger.debug("fetch: RPC failure", exc_info=True)
|
|
112
|
+
raise typer.Exit(3)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _step_sync(store: Store, rpc: RpcClient, cfg: Config, to_block: Optional[int]) -> str:
|
|
116
|
+
"""Sync feedback/revocation/response events up to ``to_block`` (or head)."""
|
|
117
|
+
n = base.sync_feedback(store, rpc, chunk=cfg.chunk_blocks, end_block=to_block, confirmations=cfg.confirmations)
|
|
118
|
+
return f"feedback rows added: {n}"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _step_timestamps(store: Store, rpc: RpcClient, batch_state: Optional[base.BatchState] = None) -> str:
|
|
122
|
+
"""Resolve block timestamps for every block referenced by feedback rows."""
|
|
123
|
+
n = base.fill_block_timestamps(store, rpc, state=batch_state)
|
|
124
|
+
still_missing = store.n_missing_block_ts()
|
|
125
|
+
if still_missing > 0:
|
|
126
|
+
typer.echo(f"WARNING: {still_missing} block(s) still missing a cached timestamp")
|
|
127
|
+
return f"block timestamps filled: {n}"
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _step_owners(store: Store, rpc: RpcClient, log_every: int = OWNER_LOG_EVERY,
|
|
131
|
+
batch_size: int = OWNER_BATCH_DEFAULT, batch_state: Optional[base.BatchState] = None) -> str:
|
|
132
|
+
"""Resolve the IdentityRegistry owner for every not-yet-resolved agent id.
|
|
133
|
+
|
|
134
|
+
Owners are resolved ``batch_size`` at a time via ``base.owners_of`` (one
|
|
135
|
+
JSON-RPC batch call per chunk instead of one ``eth_call`` per agent --
|
|
136
|
+
``owners_of`` itself falls back to per-agent calls if a chunk's batch
|
|
137
|
+
fails). Every chunk shares one ``base.BatchState`` (``batch_state``, or a
|
|
138
|
+
fresh one if not given) so that once any chunk's batch fails with an
|
|
139
|
+
``RpcBatchUnsupportedError`` (batching itself doesn't work against this
|
|
140
|
+
endpoint), every later chunk skips straight to per-agent calls instead of
|
|
141
|
+
re-attempting an identical, doomed batch call. Every agent id is
|
|
142
|
+
recorded, even when no owner resolves -- as an empty string
|
|
143
|
+
(``Store.upsert_agent_owner`` accepts one; the evidence classification
|
|
144
|
+
join already filters falsy owners out of its parties set) -- so an
|
|
145
|
+
unresolved agent is never re-queried on a later run. Logs progress at
|
|
146
|
+
INFO every ``log_every`` agents (not every chunk).
|
|
147
|
+
"""
|
|
148
|
+
agents = store.distinct_agents()
|
|
149
|
+
resolved = 0
|
|
150
|
+
batch_state = batch_state or base.BatchState()
|
|
151
|
+
for i in range(0, len(agents), batch_size):
|
|
152
|
+
chunk = agents[i:i + batch_size]
|
|
153
|
+
owners = base.owners_of(rpc, chunk, batch_size=batch_size, state=batch_state)
|
|
154
|
+
for j, agent_id in enumerate(chunk):
|
|
155
|
+
owner = owners.get(agent_id)
|
|
156
|
+
store.upsert_agent_owner(agent_id, owner or "")
|
|
157
|
+
if owner:
|
|
158
|
+
resolved += 1
|
|
159
|
+
n_done = i + j + 1
|
|
160
|
+
if log_every > 0 and n_done % log_every == 0:
|
|
161
|
+
logger.info("fetch: agent owners resolved %d/%d", n_done, len(agents))
|
|
162
|
+
return f"agent owners resolved: {resolved}/{len(agents)}"
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _step_raters(store: Store, etherscan_key: Optional[str]) -> str:
|
|
166
|
+
"""Profile not-yet-profiled raters via Etherscan, or fall back offline.
|
|
167
|
+
|
|
168
|
+
``etherscan_key``, when given, overrides ``ETHERSCAN_API_KEY`` (the key
|
|
169
|
+
itself is never echoed). Raises ``RuntimeError`` (from ``enrich_raters``)
|
|
170
|
+
when some -- but not all -- addresses failed; the caller decides how to
|
|
171
|
+
handle a partial failure.
|
|
172
|
+
"""
|
|
173
|
+
key = etherscan_key if etherscan_key is not None else os.environ.get("ETHERSCAN_API_KEY")
|
|
174
|
+
client = EtherscanClient(key) if key else None
|
|
175
|
+
# store.distinct_clients() computed exactly once here and handed to
|
|
176
|
+
# enrich_raters(addresses=...) below, instead of letting it re-query the
|
|
177
|
+
# store itself.
|
|
178
|
+
targets = store.distinct_clients()
|
|
179
|
+
if client is not None and targets:
|
|
180
|
+
eta_h = estimate_seconds(len(targets), DEFAULT_RPS) / 3600
|
|
181
|
+
typer.echo(f"profiling {len(targets)} rater(s) via Etherscan, ETA ~{eta_h:.1f}h")
|
|
182
|
+
n = enrich_raters(store, client, addresses=targets)
|
|
183
|
+
mode = store.get_sync("rater_profile_mode") or "unknown"
|
|
184
|
+
fallback_note = "" if client is not None else " (fallback: no ETHERSCAN_API_KEY; rows not written)"
|
|
185
|
+
return f"raters profiled: {n}{fallback_note}; rater profile mode: {mode}"
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _step_evidence(store: Store, rpc: RpcClient, workers: int) -> str:
|
|
189
|
+
"""Classify every not-yet-cached evidence URI referenced by feedback rows,
|
|
190
|
+
fetching+classifying up to ``workers`` URIs concurrently."""
|
|
191
|
+
n = classify_all(store, tx_parties=lambda h: base.tx_parties(rpc, h), workers=workers)
|
|
192
|
+
return f"evidence URIs classified: {n}"
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
@app.command()
|
|
196
|
+
def fetch(
|
|
197
|
+
db: Path = typer.Option(..., help="SQLite store path."),
|
|
198
|
+
to_block: Optional[int] = typer.Option(
|
|
199
|
+
None, help="Sync feedback up to this block (default: chain head - confirmations)."),
|
|
200
|
+
chunk: int = typer.Option(Config().chunk_blocks, min=1, help="Blocks per eth_getLogs call."),
|
|
201
|
+
confirmations: int = typer.Option(
|
|
202
|
+
Config().confirmations, min=0, help="Blocks to lag behind the chain head before ingesting."),
|
|
203
|
+
rpc_url: List[str] = typer.Option(
|
|
204
|
+
[], "--rpc-url", help="RPC endpoint (repeatable); overrides the default endpoint list."),
|
|
205
|
+
etherscan_key: Optional[str] = typer.Option(
|
|
206
|
+
None, "--etherscan-key", callback=_no_blank,
|
|
207
|
+
help="Etherscan API key. Prefer the ETHERSCAN_API_KEY environment variable instead: "
|
|
208
|
+
"a command-line argument is visible to other local users (e.g. via `ps`)."),
|
|
209
|
+
skip_owners: bool = typer.Option(False, "--skip-owners", help="Skip agent owner resolution."),
|
|
210
|
+
owner_batch: int = typer.Option(
|
|
211
|
+
OWNER_BATCH_DEFAULT, "--owner-batch", min=1,
|
|
212
|
+
help="Agents resolved per JSON-RPC batch call for owner resolution (1 = one call per agent)."),
|
|
213
|
+
skip_raters: bool = typer.Option(False, "--skip-raters", help="Skip rater profile enrichment."),
|
|
214
|
+
skip_evidence: bool = typer.Option(False, "--skip-evidence", help="Skip evidence URI classification."),
|
|
215
|
+
evidence_workers: int = typer.Option(
|
|
216
|
+
8, "--evidence-workers", min=1,
|
|
217
|
+
help="Concurrent worker threads used to fetch+classify evidence URIs."),
|
|
218
|
+
dry_run: bool = typer.Option(
|
|
219
|
+
False, "--dry-run", help="Print what would be synced and exit, without any network calls."),
|
|
220
|
+
) -> None:
|
|
221
|
+
"""Sync Base ERC-8004 feedback, block timestamps, agent owners, rater
|
|
222
|
+
profiles and evidence levels into the SQLite store at --db.
|
|
223
|
+
|
|
224
|
+
Exits 0 on full success; 1 for a bad-input error (an invalid option
|
|
225
|
+
combination that Config itself rejects); 2 when rater profiling partially
|
|
226
|
+
failed (some addresses errored, the rest -- and every other step -- still
|
|
227
|
+
completed; see _step_raters); 3 when an RPC call failed on every retry.
|
|
228
|
+
--dry-run never opens or creates the store file when it does not already
|
|
229
|
+
exist.
|
|
230
|
+
"""
|
|
231
|
+
if dry_run:
|
|
232
|
+
if db.exists():
|
|
233
|
+
with Store(db) as store:
|
|
234
|
+
last = store.get_sync("last_block")
|
|
235
|
+
start = int(last) + 1 if last is not None else base.DEPLOY_BLOCK
|
|
236
|
+
else:
|
|
237
|
+
start = base.DEPLOY_BLOCK
|
|
238
|
+
end = to_block if to_block is not None else "head"
|
|
239
|
+
typer.echo(f"would sync from block {start} to {end}")
|
|
240
|
+
raise typer.Exit(0)
|
|
241
|
+
|
|
242
|
+
try:
|
|
243
|
+
cfg = Config(chunk_blocks=chunk, confirmations=confirmations,
|
|
244
|
+
rpc_urls=tuple(rpc_url) if rpc_url else Config().rpc_urls)
|
|
245
|
+
except ValueError as e:
|
|
246
|
+
typer.echo(f"ERROR: {e}")
|
|
247
|
+
raise typer.Exit(1)
|
|
248
|
+
|
|
249
|
+
rpc = RpcClient(cfg.rpc_urls, user_agent=cfg.user_agent)
|
|
250
|
+
partial_failure = False
|
|
251
|
+
# Shared across every batch-capable step of this run: once one step's
|
|
252
|
+
# batch calls are found to fail structurally (or get batch-rate-limited),
|
|
253
|
+
# every later step also skips straight to per-item calls instead of
|
|
254
|
+
# re-discovering the same doomed batch failure from scratch.
|
|
255
|
+
batch_state = base.BatchState()
|
|
256
|
+
|
|
257
|
+
with Store(db) as store:
|
|
258
|
+
with _rpc_guard():
|
|
259
|
+
typer.echo(_step_sync(store, rpc, cfg, to_block))
|
|
260
|
+
typer.echo(_step_timestamps(store, rpc, batch_state=batch_state))
|
|
261
|
+
if skip_owners:
|
|
262
|
+
typer.echo("agent owner resolution skipped")
|
|
263
|
+
else:
|
|
264
|
+
typer.echo(_step_owners(store, rpc, batch_size=owner_batch, batch_state=batch_state))
|
|
265
|
+
|
|
266
|
+
if skip_raters:
|
|
267
|
+
mode = store.get_sync("rater_profile_mode") or "unknown"
|
|
268
|
+
typer.echo(f"raters profiling skipped; rater profile mode: {mode}")
|
|
269
|
+
else:
|
|
270
|
+
try:
|
|
271
|
+
typer.echo(_step_raters(store, etherscan_key))
|
|
272
|
+
except RuntimeError as e:
|
|
273
|
+
typer.echo(f"WARNING: {e}")
|
|
274
|
+
partial_failure = True
|
|
275
|
+
|
|
276
|
+
with _rpc_guard():
|
|
277
|
+
if skip_evidence:
|
|
278
|
+
typer.echo("evidence classification skipped")
|
|
279
|
+
else:
|
|
280
|
+
typer.echo(_step_evidence(store, rpc, workers=evidence_workers))
|
|
281
|
+
|
|
282
|
+
if partial_failure:
|
|
283
|
+
raise typer.Exit(2)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
@app.command()
|
|
287
|
+
def score(
|
|
288
|
+
db: Path = typer.Option(..., help="SQLite store path."),
|
|
289
|
+
out: Path = typer.Option(Path("scores.csv"), help="CSV path to write results to."),
|
|
290
|
+
bootstrap_n: int = typer.Option(
|
|
291
|
+
Config().bootstrap_n, min=0, help="Bootstrap resamples for the confidence interval."),
|
|
292
|
+
min_clusters: int = typer.Option(
|
|
293
|
+
Config().min_clusters, min=1, help="Minimum distinct rater clusters required to score a ratee."),
|
|
294
|
+
) -> None:
|
|
295
|
+
"""Compute robust reputation scores for every ratee in the store and
|
|
296
|
+
write them to --out as a CSV with RESULT_COLUMNS columns.
|
|
297
|
+
|
|
298
|
+
Exits 1 with "no records" if the store has no feedback rows yet (run
|
|
299
|
+
fetch first), or with an ERROR line if the given options or the data
|
|
300
|
+
itself is rejected (e.g. too many candidate sybil pairs, or malformed
|
|
301
|
+
records).
|
|
302
|
+
"""
|
|
303
|
+
try:
|
|
304
|
+
cfg = Config(bootstrap_n=bootstrap_n, min_clusters=min_clusters)
|
|
305
|
+
except ValueError as e:
|
|
306
|
+
typer.echo(f"ERROR: {e}")
|
|
307
|
+
raise typer.Exit(1)
|
|
308
|
+
|
|
309
|
+
with Store(db) as store:
|
|
310
|
+
records = store.load_records()
|
|
311
|
+
if records.empty:
|
|
312
|
+
typer.echo("no records")
|
|
313
|
+
raise typer.Exit(1)
|
|
314
|
+
|
|
315
|
+
try:
|
|
316
|
+
clusters = cluster_raters(profiles_from_records(records, store.load_rater_meta()), cfg)
|
|
317
|
+
result = score_fn(records, cfg, clusters=clusters)
|
|
318
|
+
except ValueError as e:
|
|
319
|
+
typer.echo(f"ERROR: {e}")
|
|
320
|
+
raise typer.Exit(1)
|
|
321
|
+
|
|
322
|
+
mode = store.get_sync("rater_profile_mode") or "unknown"
|
|
323
|
+
|
|
324
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
325
|
+
result.to_csv(out, index=False)
|
|
326
|
+
|
|
327
|
+
n_scored = int((result["insufficient"] == 0).sum())
|
|
328
|
+
n_insufficient = int((result["insufficient"] == 1).sum())
|
|
329
|
+
n_sybil = int(result["sybil_flag"].sum())
|
|
330
|
+
typer.echo(
|
|
331
|
+
f"scored {n_scored} ratee(s), {n_insufficient} insufficient, {n_sybil} sybil-flagged "
|
|
332
|
+
f"-> {out} (rater profile mode: {mode})")
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
# Figure titles (heading text) -> file names, in report order. Titles are what
|
|
336
|
+
# `render_markdown` uses as the Markdown heading/alt text (never a raw
|
|
337
|
+
# filename); the mapping keeps the two in one place.
|
|
338
|
+
_FIGURES = {
|
|
339
|
+
"Fig 1. Mean vs robust score": "fig1_mean_vs_robust.png",
|
|
340
|
+
"Fig 2. Biggest rank drops": "fig2_rank_shift.png",
|
|
341
|
+
"Fig 3. Evidence levels": "fig3_evidence.png",
|
|
342
|
+
"Fig 4. Largest rater clusters": "fig4_sybil_clusters.png",
|
|
343
|
+
"Fig 5. Ranking stability": "fig5_sensitivity.png",
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _evidence_shares_for_provenance(records: pandas.DataFrame) -> dict:
|
|
348
|
+
"""Per-record evidence-level shares (0..3) among non-revoked records, as a
|
|
349
|
+
JSON-friendly ``{level: share}`` dict -- the same figure the report's
|
|
350
|
+
headline numbers are built from (see ``robustrep.report.render``)."""
|
|
351
|
+
non_revoked = records[records["revoked"] == 0] if "revoked" in records.columns else records
|
|
352
|
+
shares = evidence_level_shares(non_revoked)
|
|
353
|
+
return {int(level): float(share) for level, share in shares.items()}
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _provenance(mode: str, cfg: Config, records: pandas.DataFrame) -> dict:
|
|
357
|
+
config = dict(
|
|
358
|
+
bootstrap_n=cfg.bootstrap_n,
|
|
359
|
+
bootstrap_seed=cfg.bootstrap_seed,
|
|
360
|
+
min_clusters=cfg.min_clusters,
|
|
361
|
+
evidence_weights=list(cfg.evidence_weights),
|
|
362
|
+
sybil_jaccard=cfg.sybil_jaccard,
|
|
363
|
+
sybil_window_s=cfg.sybil_window_s,
|
|
364
|
+
sybil_max_group=cfg.sybil_max_group,
|
|
365
|
+
sybil_flag_share=cfg.sybil_flag_share,
|
|
366
|
+
evidence_level_shares=_evidence_shares_for_provenance(records),
|
|
367
|
+
)
|
|
368
|
+
return dict(
|
|
369
|
+
rater_profile_mode=mode,
|
|
370
|
+
confirmations=cfg.confirmations,
|
|
371
|
+
config=config,
|
|
372
|
+
versions={"robustrep": __version__, "numpy": numpy.__version__, "pandas": pandas.__version__},
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _write_report(target: Path, result, records, clusters, sens, adv, block: int, provenance: dict,
|
|
377
|
+
top_n: int) -> None:
|
|
378
|
+
"""Draw all 5 figures and write report.md + scores.json into `target`."""
|
|
379
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
380
|
+
fig_mean_vs_robust(result, out=target / _FIGURES["Fig 1. Mean vs robust score"])
|
|
381
|
+
fig_rank_shift(result, out=target / _FIGURES["Fig 2. Biggest rank drops"], top_n=top_n)
|
|
382
|
+
fig_evidence(records, out=target / _FIGURES["Fig 3. Evidence levels"])
|
|
383
|
+
fig_sybil_clusters(records, clusters, out=target / _FIGURES["Fig 4. Largest rater clusters"])
|
|
384
|
+
fig_sensitivity(sens, out=target / _FIGURES["Fig 5. Ranking stability"])
|
|
385
|
+
md = render_markdown(result, records, block=block, figures=_FIGURES, sensitivity=sens,
|
|
386
|
+
adversarial=adv, provenance=provenance)
|
|
387
|
+
(target / "report.md").write_text(md)
|
|
388
|
+
export_json(result, block=block, out=target / "scores.json", config=provenance.get("config"))
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _publish_latest(out_dir: Path, block_dir: Path) -> None:
|
|
392
|
+
"""Publish a copy of `block_dir` as `out_dir/latest/`.
|
|
393
|
+
|
|
394
|
+
Near-atomic: the old `latest/` is removed then the staged copy is moved in
|
|
395
|
+
-- a reader may briefly see no `latest/` at all, in the gap between the
|
|
396
|
+
removal and the move. Builds the new contents in a temp directory first, so
|
|
397
|
+
that gap is as short as a single `os.replace` (moving the fully-staged
|
|
398
|
+
directory into place, no partial writes ever visible) rather than however
|
|
399
|
+
long the figures/report.md/scores.json themselves take to generate; stale
|
|
400
|
+
files from an earlier run never linger alongside the new ones.
|
|
401
|
+
"""
|
|
402
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
403
|
+
latest = out_dir / "latest"
|
|
404
|
+
tmp_parent = Path(tempfile.mkdtemp(dir=out_dir))
|
|
405
|
+
try:
|
|
406
|
+
staged = tmp_parent / "latest"
|
|
407
|
+
shutil.copytree(block_dir, staged)
|
|
408
|
+
if latest.exists():
|
|
409
|
+
shutil.rmtree(latest)
|
|
410
|
+
os.replace(staged, latest)
|
|
411
|
+
finally:
|
|
412
|
+
shutil.rmtree(tmp_parent, ignore_errors=True)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
@app.command()
|
|
416
|
+
def report(
|
|
417
|
+
db: Path = typer.Option(..., help="SQLite store path."),
|
|
418
|
+
out_dir: Path = typer.Option(Path("reports"), help="Output directory."),
|
|
419
|
+
bootstrap_n: int = typer.Option(
|
|
420
|
+
1000, min=0, help="Bootstrap resamples for the confidence interval."),
|
|
421
|
+
top_n: int = typer.Option(
|
|
422
|
+
100, min=1, help="Top-N ratees (by naive mean / robust score) considered for the "
|
|
423
|
+
"rank-shift and sensitivity figures."),
|
|
424
|
+
) -> None:
|
|
425
|
+
"""Generate figures 1-5, report.md and scores.json under out_dir/<block>/,
|
|
426
|
+
then atomically publish a copy to out_dir/latest/ (a fixed link always has
|
|
427
|
+
the latest report while every past run stays addressable by block number).
|
|
428
|
+
|
|
429
|
+
Exits 1 with "no records" if the store has no feedback rows yet (run
|
|
430
|
+
fetch first), or with an ERROR line if the given options or the data
|
|
431
|
+
itself is rejected (e.g. too many candidate sybil pairs, or malformed
|
|
432
|
+
records).
|
|
433
|
+
"""
|
|
434
|
+
try:
|
|
435
|
+
cfg = Config(bootstrap_n=bootstrap_n)
|
|
436
|
+
except ValueError as e:
|
|
437
|
+
typer.echo(f"ERROR: {e}")
|
|
438
|
+
raise typer.Exit(1)
|
|
439
|
+
|
|
440
|
+
try:
|
|
441
|
+
with Store(db) as store:
|
|
442
|
+
records = store.load_records()
|
|
443
|
+
if records.empty:
|
|
444
|
+
typer.echo("no records")
|
|
445
|
+
raise typer.Exit(1)
|
|
446
|
+
meta = store.load_rater_meta()
|
|
447
|
+
clusters = cluster_raters(profiles_from_records(records, meta), cfg)
|
|
448
|
+
result = score_fn(records, cfg, clusters=clusters)
|
|
449
|
+
mode = store.get_sync("rater_profile_mode") or "unknown"
|
|
450
|
+
last_block = store.get_sync("last_block")
|
|
451
|
+
except ValueError as e:
|
|
452
|
+
typer.echo(f"ERROR: {e}")
|
|
453
|
+
raise typer.Exit(1)
|
|
454
|
+
block = int(last_block) if last_block is not None else 0
|
|
455
|
+
|
|
456
|
+
sens = sensitivity_table(records, cfg, meta=meta, top_n=top_n)
|
|
457
|
+
adv = scenario_table()
|
|
458
|
+
provenance = _provenance(mode, cfg, records)
|
|
459
|
+
|
|
460
|
+
block_dir = out_dir / str(block)
|
|
461
|
+
_write_report(block_dir, result, records, clusters, sens, adv, block, provenance, top_n)
|
|
462
|
+
_publish_latest(out_dir, block_dir)
|
|
463
|
+
|
|
464
|
+
typer.echo(f"report written to {block_dir} and {out_dir / 'latest'} "
|
|
465
|
+
f"(rater profile mode: {mode})")
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
if __name__ == "__main__":
|
|
469
|
+
app()
|
robustrep/config.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Tunable parameters. Defaults are provisional until the sensitivity analysis in the report confirms them."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
# Blocks to lag behind the chain head before ingesting: guards against a re-org
|
|
7
|
+
# near the tip leaving phantom feedback rows behind. Single source of truth for
|
|
8
|
+
# both Config.confirmations' default and robustrep.sources.base_erc8004.sync_feedback's
|
|
9
|
+
# default, so the two can never silently drift apart.
|
|
10
|
+
DEFAULT_CONFIRMATIONS = 20
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class Config:
|
|
15
|
+
# evidence level 0..3 -> weight
|
|
16
|
+
evidence_weights: tuple[float, float, float, float] = (0.1, 0.3, 0.7, 1.0)
|
|
17
|
+
# sybil clustering
|
|
18
|
+
sybil_window_s: int = 24 * 3600
|
|
19
|
+
sybil_jaccard: float = 0.8
|
|
20
|
+
sybil_max_group: int = 2000 # max raters per ratee considered when generating candidate sybil pairs
|
|
21
|
+
sybil_max_pairs: int = 5_000_000 # global cap on candidate pairs; cluster_raters fails fast past this
|
|
22
|
+
sybil_flag_share: float = 0.5
|
|
23
|
+
# aggregation
|
|
24
|
+
min_clusters: int = 3
|
|
25
|
+
bootstrap_n: int = 1000
|
|
26
|
+
bootstrap_seed: int = 0
|
|
27
|
+
ci_level: float = 0.95
|
|
28
|
+
# data source
|
|
29
|
+
rpc_urls: tuple[str, ...] = ("https://mainnet.base.org",)
|
|
30
|
+
chunk_blocks: int = 2000
|
|
31
|
+
user_agent: str = "robustrep/0.1 (+https://github.com/iamdongwang/robustrep)"
|
|
32
|
+
confirmations: int = DEFAULT_CONFIRMATIONS
|
|
33
|
+
|
|
34
|
+
def __post_init__(self) -> None:
|
|
35
|
+
if len(self.evidence_weights) != 4:
|
|
36
|
+
raise ValueError("evidence_weights must have exactly 4 entries")
|
|
37
|
+
if any(w <= 0 for w in self.evidence_weights):
|
|
38
|
+
raise ValueError("evidence_weights must be > 0 (zero weights make bootstrap resamples degenerate)")
|
|
39
|
+
if not 0 < self.ci_level < 1:
|
|
40
|
+
raise ValueError("ci_level must be in (0, 1)")
|
|
41
|
+
if self.bootstrap_n < 0:
|
|
42
|
+
raise ValueError("bootstrap_n must be >= 0")
|
|
43
|
+
if self.min_clusters < 1:
|
|
44
|
+
raise ValueError("min_clusters must be >= 1")
|
|
45
|
+
if not 0 < self.sybil_jaccard <= 1:
|
|
46
|
+
raise ValueError("sybil_jaccard must be in (0, 1]") # 0 would match every pair's Jaccard signal
|
|
47
|
+
if self.sybil_max_pairs < 1:
|
|
48
|
+
raise ValueError("sybil_max_pairs must be >= 1")
|
|
49
|
+
if not 0 <= self.sybil_flag_share <= 1:
|
|
50
|
+
raise ValueError("sybil_flag_share must be in [0, 1]")
|
|
51
|
+
if self.sybil_window_s < 0:
|
|
52
|
+
raise ValueError("sybil_window_s must be >= 0")
|
|
53
|
+
if self.chunk_blocks < 1:
|
|
54
|
+
raise ValueError("chunk_blocks must be >= 1")
|
|
55
|
+
if len(self.rpc_urls) < 1:
|
|
56
|
+
raise ValueError("rpc_urls must have at least 1 entry")
|
|
57
|
+
if self.confirmations < 0:
|
|
58
|
+
raise ValueError("confirmations must be >= 0")
|
robustrep/evidence.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Evidence level (0..3) for a rating and the level -> weight mapping."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import re
|
|
6
|
+
from typing import Callable, Optional
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from .config import Config
|
|
11
|
+
|
|
12
|
+
TX_RE = re.compile(r"0x[0-9a-fA-F]{64}(?![0-9a-fA-F])")
|
|
13
|
+
# Deliberately strict (double-quoted JSON keys only): misses fall to level 1,
|
|
14
|
+
# the conservative direction.
|
|
15
|
+
TASK_KEY_RE = re.compile(r'"(?:taskId|task_id|jobId|job_id|orderId|order_id)"\s*:')
|
|
16
|
+
|
|
17
|
+
FetchText = Callable[[str], Optional[str]]
|
|
18
|
+
TxParties = Callable[[str], Optional[set[str]]] # tx hash -> {from, to} lowercased, or None
|
|
19
|
+
|
|
20
|
+
_log = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def weights_for(levels: pd.Series, cfg: Config) -> pd.Series:
|
|
24
|
+
"""Map evidence levels (0..3) to weights per cfg.evidence_weights.
|
|
25
|
+
|
|
26
|
+
Preserves the input index. Raises ValueError if any level is outside
|
|
27
|
+
[0, 1, 2, 3] or null.
|
|
28
|
+
"""
|
|
29
|
+
bad = sorted(set(levels.dropna()) - {0, 1, 2, 3})
|
|
30
|
+
if bad:
|
|
31
|
+
raise ValueError(f"evidence_level values {bad} not in [0, 1, 2, 3]")
|
|
32
|
+
if levels.isna().any():
|
|
33
|
+
raise ValueError("evidence_level contains null")
|
|
34
|
+
w = cfg.evidence_weights
|
|
35
|
+
return levels.astype(int).map(lambda l: w[l]).astype(float)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _verified(h: str, tx_parties: TxParties, parties_lc: set[str]) -> bool:
|
|
39
|
+
try:
|
|
40
|
+
found = tx_parties(h)
|
|
41
|
+
return bool(found) and bool({a.lower() for a in found if a} & parties_lc)
|
|
42
|
+
except Exception:
|
|
43
|
+
_log.debug("tx_parties(%s) failed; treating as unverified", h, exc_info=True)
|
|
44
|
+
return False
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def classify(uri: Optional[str], fetch_text: FetchText, tx_parties: TxParties, parties: set[str]) -> int:
|
|
48
|
+
"""Evidence level 0..3 per the spec table.
|
|
49
|
+
|
|
50
|
+
parties: lowercased addresses of rater and ratee; falsy entries ignored.
|
|
51
|
+
|
|
52
|
+
Callable contracts: `fetch_text` exceptions propagate to the caller (its
|
|
53
|
+
own fetcher must handle its own errors); `tx_parties` exceptions, or a
|
|
54
|
+
None/falsy return, are tolerated and degrade that hash to unverified
|
|
55
|
+
(logged at DEBUG), never raising out of `classify`.
|
|
56
|
+
"""
|
|
57
|
+
if not uri or not uri.strip():
|
|
58
|
+
return 0
|
|
59
|
+
text = fetch_text(uri)
|
|
60
|
+
if text is None:
|
|
61
|
+
return 1
|
|
62
|
+
hashes = TX_RE.findall(text)
|
|
63
|
+
if not hashes and not TASK_KEY_RE.search(text):
|
|
64
|
+
return 1
|
|
65
|
+
parties_lc = {p.lower() for p in parties if p}
|
|
66
|
+
for h in dict.fromkeys(x.lower() for x in hashes):
|
|
67
|
+
if _verified(h, tx_parties, parties_lc):
|
|
68
|
+
return 3
|
|
69
|
+
return 2
|