factblock 1.0.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
factblock/__init__.py ADDED
@@ -0,0 +1,30 @@
1
+ """FactBlock table format: reference reader and validator (SPEC.md).
2
+
3
+ scan(bundle, as_of, valid_at=None) -> Scan # .nodes/.edges/.resolutions Arrow tables, .certificate
4
+ validate(bundle) -> list[Check] # one row per check id in SPEC section 2
5
+ resolve(bundle, fact_key, as_of, valid_at=None, rules_as_of=None) -> dict # SPEC 4.4
6
+ write_parquet(bundle, out) -> Path # JSONL bundle to the Parquet profile, SPEC 5.3
7
+ extract(text, observed_at, ...) -> dict # the claims profile through a model you choose (extract.py)
8
+ write_bundle(result, out, append=False) # dicts to a JSONL bundle, growing one batch at a time
9
+ recall(bundle, query, as_of, valid_at=None, limit=10) -> dict # the blocks about something, as of an instant
10
+ context(bundle, query, as_of) -> str # the same recall as prompt lines
11
+ why(bundle, node_id, as_of, valid_at=None, depth=3) -> dict # the chain behind one block, SPEC 4.5
12
+ leak(bundle, questions) -> dict # answers that rest on blocks learned after the question was asked
13
+ sync(bundle, store, as_of=None, push=True, pull=True) -> dict # folder <-> a store (TckgStore, or yours) by identity
14
+ to_claimreview(bundle, as_of, base_url=None) -> dict # verdicts as schema.org ClaimReview JSON-LD, SPEC 9.2
15
+ to_okf(bundle, as_of, out) -> Path # one OKF concept document per visible node, SPEC 9.1
16
+ """
17
+ from .bundle import Bundle, write_bundle
18
+ from .claimreview import to_claimreview
19
+ from .okf import to_okf
20
+ from .extract import extract, load_profile
21
+ from .leak import leak
22
+ from .parquet import write_parquet
23
+ from .recall import context, recall
24
+ from .resolve import resolve
25
+ from .scan import Scan, scan
26
+ from .sync import Store, TckgStore, sync
27
+ from .validate import Check, validate
28
+ from .why import why
29
+
30
+ __all__ = ["Bundle", "Scan", "scan", "Check", "validate", "resolve", "write_parquet", "extract", "load_profile", "write_bundle", "why", "leak", "sync", "Store", "TckgStore", "to_claimreview", "to_okf", "recall", "context"]
factblock/__main__.py ADDED
@@ -0,0 +1,237 @@
1
+ """The factblock command. `factblock --help` lists the subcommands in the order you meet them:
2
+ sample, extract, scan, recall, why, resolve, leak, validate, sync, then the projections and adapters.
3
+ Reads print for people; `--json` prints the same answer as JSON."""
4
+ import argparse
5
+ import json
6
+ import os
7
+ import pathlib
8
+ import shutil
9
+ import sys
10
+
11
+ from . import Bundle, TckgStore, extract, leak, recall, resolve, scan, sync, to_claimreview, to_okf, validate, why, write_bundle, write_parquet
12
+ from .adapters.factcheck import bundle_from_factcheck, search as factcheck_search
13
+
14
+ # the wheel carries samples/rates at factblock/samples/rates (pyproject force-include); a checkout has it at the repo root
15
+ SAMPLES = next((d for d in (pathlib.Path(__file__).parent / "samples", pathlib.Path(__file__).resolve().parents[1] / "samples") if (d / "rates").exists()), None)
16
+
17
+
18
+ def _day(v):
19
+ return v.date().isoformat() if hasattr(v, "date") else str(v)[:10]
20
+
21
+
22
+ def _cert(c):
23
+ m = c.get("masked", {}); b = c.get("backfill")
24
+ parts = [f"as of {c['as_of'][:10]}"]
25
+ if c.get("valid_at", "")[:10] != c["as_of"][:10]:
26
+ parts.append(f"valid at {c['valid_at'][:10]}")
27
+ parts.append("hidden: " + (", ".join(f"{n} {k}{'s' if n != 1 else ''}" for k, n in m.items()) if m else "nothing"))
28
+ if b:
29
+ parts.append(f"backfilled: {b['rows']} rows in {b['batches']} batch{'es' if b['batches'] != 1 else ''}")
30
+ return " ".join(parts)
31
+
32
+
33
+ def _print_scan(r):
34
+ print(_cert(r.certificate))
35
+ nodes = r.nodes.to_pylist()
36
+ for n in sorted(nodes, key=lambda n: (n["asserted_at"], n["id"])):
37
+ tail = f" [superseded by {n['superseded_by']}]" if n.get("superseded_by") else ""
38
+ print(f"{n['id']:<12} {n['kind']:<11} {_day(n['asserted_at'])} {n.get('statement') or ''}{tail}")
39
+ if r.edges.num_rows:
40
+ print("edges:")
41
+ for e in sorted(r.edges.to_pylist(), key=lambda e: (e["asserted_at"], e["source_id"])):
42
+ print(f" {e['source_id']} --{e['edge_type']}--> {e['target_id']} {_day(e['asserted_at'])}")
43
+ if r.resolutions.num_rows:
44
+ print("verdicts:")
45
+ for v in sorted(r.resolutions.to_pylist(), key=lambda v: v["decided_at"]):
46
+ print(f" {v['target_id']:<12} {v.get('outcome') or v.get('value')} decided {_day(v['decided_at'])}" + (f" by {v['resolver']}" if v.get("resolver") else ""))
47
+
48
+
49
+ def _print_why(r):
50
+ if not r["chain"]:
51
+ print(f"{r['root']}: {r.get('reason', 'no chain')} {_cert(r['certificate'])}")
52
+ return
53
+ for row in sorted(r["chain"], key=lambda x: (len(x["path"]), x["path"])):
54
+ via = f" ({row['via']['edge_type']})" if row["via"] else ""
55
+ print(" " * row["depth"] + f"{row['role']}{via}: {row['id']} {_day(row['asserted_at'])} {row['statement'] or ''}")
56
+ print(_cert(r["certificate"]))
57
+
58
+
59
+ def _print_resolve(k, r):
60
+ if r["status"] == "answered":
61
+ print(f"{k}: {json.dumps(r['value'])} policy {r.get('policy')}, {len(r['candidates'])} candidate{'s' if len(r['candidates']) != 1 else ''}")
62
+ else:
63
+ print(f"{k}: no answer, {r['reason']}" + (f" since {r['unresolved_since'][:10]}" if r.get("unresolved_since") else ""))
64
+ for c in r.get("candidates", []):
65
+ print(f" {c['id']:<12} {json.dumps(c['value'])} valid from {c['valid_from'][:10]} known {c['known_at'][:10]}")
66
+ print(_cert(r["certificate"]))
67
+
68
+
69
+ def main():
70
+ p = argparse.ArgumentParser(prog="factblock", description="Agent memory for decisions: dated claims and their causal links, in a folder, read as of any instant.")
71
+ sub = p.add_subparsers(dest="cmd", metavar="command")
72
+
73
+ def cmd(name, help_, as_of=True, valid_at=True, json_=True):
74
+ s = sub.add_parser(name, help=help_, description=help_)
75
+ s.add_argument("bundle", help="the bundle directory")
76
+ if as_of:
77
+ s.add_argument("--as-of", required=True, help="the instant to read as of (date or ISO 8601); nothing learned later is shown")
78
+ if valid_at:
79
+ s.add_argument("--valid-at", help="the instant the content must hold at; default: as-of")
80
+ if json_:
81
+ s.add_argument("--json", action="store_true", help="print the answer as JSON")
82
+ return s
83
+
84
+ s = sub.add_parser("sample", help="copy the sample bundle (six dated claims, a reversal, three verdicts) into a folder", description="copy the sample bundle into a folder")
85
+ s.add_argument("out", help="folder to create, e.g. brain/")
86
+ e = sub.add_parser("extract", help="text in, dated blocks and their links appended to a bundle", description="text in, dated blocks and their links appended to a bundle")
87
+ e.add_argument("source", help="a text file, or - for stdin")
88
+ e.add_argument("--observed-at", required=True, help="when it was said or written (ISO 8601)")
89
+ e.add_argument("-o", "--out", required=True, help="bundle directory; created or appended to")
90
+ e.add_argument("--speaker", help="who said it, as written")
91
+ e.add_argument("--source-name", dest="source_name", help="where it came from (a channel, a document)")
92
+ e.add_argument("--provider", default="gemini", choices=["gemini", "openai", "fake"], help="gemini (GEMINI_API_KEY or Vertex), openai (OPENAI_API_KEY), or fake: no model, sentence split, for trying the format")
93
+ e.add_argument("--model", help="model name for the provider")
94
+ e.add_argument("--known-at", help="when you learned it; default now")
95
+ e.add_argument("--backfill", action="store_true", help="known_at = observed_at: material from the past, known when it was said")
96
+ e.add_argument("--namespace", default="local")
97
+ cmd("scan", "what the folder knew as of an instant: blocks, edges, verdicts, and what was hidden")
98
+ rc = cmd("recall", "the blocks about something as of an instant, ranked; the read an agent makes before it answers")
99
+ rc.add_argument("query", help="words to look for in statements, quotes and speakers")
100
+ rc.add_argument("--limit", type=int, default=10)
101
+ rc.add_argument("--all-kinds", action="store_true", help="include entities and episodes, not only claims and predictions")
102
+ w = cmd("why", "the chain behind one block as of an instant: causes, effects, successors, contradictions")
103
+ w.add_argument("node_id", help="the block id (see scan)")
104
+ w.add_argument("--depth", type=int, default=3, help="hops to walk; default 3")
105
+ r = cmd("resolve", "one value for a declared fact, by its declared policy, or the reason there is none")
106
+ r.add_argument("fact_key", help="the declared fact, e.g. belief:fed:direction")
107
+ r.add_argument("--rules-as-of", help="apply the declaration in force at this instant instead of as-of")
108
+ k = cmd("leak", "which answers in a dated question set rest on blocks learned after the question was asked", as_of=False, valid_at=False, json_=False)
109
+ k.add_argument("questions", help="JSONL: {id?, asked_at, evidence: [block id, ...]} per line")
110
+ cmd("validate", "the conformance checks of SPEC.md section 2; exit 1 if any fails", as_of=False, valid_at=False, json_=False)
111
+ y = sub.add_parser("sync", help="folder <-> a hosted ledger, both ways, by identity; known_at travels as batches", description="folder <-> a hosted ledger (tckg), both ways")
112
+ y.add_argument("bundle")
113
+ y.add_argument("url", help="the ledger, e.g. https://tckg.factagora.com")
114
+ y.add_argument("--space", help="whose memory this is on the ledger; rows that carry their own space keep it")
115
+ y.add_argument("--token", help="API key of the tenant; default $TCKG_TOKEN")
116
+ y.add_argument("--as-of", help="pull rows the ledger knew by this instant; default now")
117
+ y.add_argument("--push-only", action="store_true")
118
+ y.add_argument("--pull-only", action="store_true")
119
+ cr = cmd("to-claimreview", "verdicts visible as of an instant, as schema.org ClaimReview JSON-LD", json_=False)
120
+ cr.add_argument("--base-url", help="each review's url becomes <base-url>/<block id>")
121
+ ok = cmd("to-okf", "the blocks visible as of an instant, as an OKF bundle (markdown + frontmatter)", json_=False)
122
+ ok.add_argument("out", help="directory to write")
123
+ c = sub.add_parser("to-parquet", help="the same bundle in the Parquet profile, for DuckDB, Spark, Databricks", description="JSONL bundle to the Parquet profile")
124
+ c.add_argument("bundle"); c.add_argument("out")
125
+ fc = sub.add_parser("from-factcheck", help="published fact-checks in (Google Fact Check Tools API shape), dated claims with dated verdicts out", description="fact-checks in, dated claims with dated verdicts out")
126
+ fc.add_argument("-o", "--out", required=True, help="bundle directory; created or appended to")
127
+ fc.add_argument("--query", help="search the live API; needs --api-key or $FACTCHECK_API_KEY")
128
+ fc.add_argument("--language", help="BCP-47 filter for --query, e.g. en or ko")
129
+ fc.add_argument("--pages", type=int, default=1, help="pages of 100 to fetch for --query")
130
+ fc.add_argument("--api-key")
131
+ fc.add_argument("--from-json", help="a saved claims:search response, or a JSON array of claims")
132
+ fc.add_argument("--namespace", default="factcheck")
133
+ a = p.parse_args()
134
+ if not a.cmd: # bare `factblock`: show what it can do, not an error
135
+ p.print_help()
136
+ print("\nStart here: factblock sample brain/ && factblock scan brain/ --as-of 2024-05-01")
137
+ return
138
+ out_json = lambda d: print(json.dumps(d, indent=1, default=str)) # noqa: E731
139
+
140
+ if a.cmd == "sample":
141
+ if pathlib.Path(a.out).exists():
142
+ p.error(f"{a.out} exists; pick a new folder")
143
+ shutil.copytree(SAMPLES / "rates", a.out)
144
+ print(f"{a.out}: the sample bundle. Try: factblock scan {a.out} --as-of 2024-05-01")
145
+ elif a.cmd == "extract":
146
+ text = sys.stdin.read() if a.source == "-" else open(a.source).read()
147
+ existing = Bundle(a.out) if (pathlib.Path(a.out) / "factblock.json").exists() else None
148
+ r = extract(text, a.observed_at, speaker=a.speaker, source=a.source_name, provider=a.provider, model=a.model,
149
+ known_at=a.known_at, backfill=a.backfill, namespace=a.namespace, existing=existing)
150
+ write_bundle(r, a.out, append=True)
151
+ s_ = r["summary"]
152
+ print(f"{a.out}: +{s_['blocks']} blocks, +{s_['entities']} entities, +{s_['links']} links"
153
+ + (f" ({s_['links_dropped']} dropped)" if s_['links_dropped'] else "") + f", batch {s_['batch']}")
154
+ elif a.cmd == "scan":
155
+ r = scan(a.bundle, a.as_of, a.valid_at)
156
+ if a.json:
157
+ out_json({"certificate": r.certificate,
158
+ "nodes": r.nodes.select([c for c in ("id", "kind", "statement", "asserted_at", "superseded_by") if c in r.nodes.column_names]).to_pylist(),
159
+ "edges": r.edges.select(["source_id", "target_id", "edge_type", "asserted_at"]).to_pylist() if r.edges.num_rows else [],
160
+ "resolutions": r.resolutions.to_pylist()})
161
+ else:
162
+ _print_scan(r)
163
+ elif a.cmd == "recall":
164
+ r = recall(a.bundle, a.query, a.as_of, a.valid_at, a.limit, kinds=() if a.all_kinds else ("claim", "prediction"))
165
+ if a.json:
166
+ out_json(r)
167
+ else:
168
+ for i in r["items"]:
169
+ print(f"{i['id']:<12} {i['kind']:<11} {_day(i['asserted_at'])} {i['statement']}" + (f" ({i['speaker']})" if i.get("speaker") else ""))
170
+ print(f"{len(r['items'])} of {r['matched']} matching {_cert(r['certificate'])}")
171
+ elif a.cmd == "why":
172
+ r = why(a.bundle, a.node_id, a.as_of, a.valid_at, a.depth)
173
+ out_json(r) if a.json else _print_why(r)
174
+ elif a.cmd == "resolve":
175
+ r = resolve(a.bundle, a.fact_key, a.as_of, a.valid_at, a.rules_as_of)
176
+ out_json(r) if a.json else _print_resolve(a.fact_key, r)
177
+ elif a.cmd == "leak":
178
+ r = leak(a.bundle, a.questions)
179
+ for q in r["per_question"]:
180
+ if q["leaked"] or q["missing"]:
181
+ print(f"{q['id']}: asked {q['asked_at'][:10]}, " + ", ".join(f"{l['id']} known {l['known_at'][:10]} ({l['reason']})" for l in q["leaked"])
182
+ + (f", missing {q['missing']}" if q["missing"] else ""))
183
+ print(f"{r['leaked_questions']}/{r['questions']} questions leak ({r['leak_rate']:.0%}), {r['leaked_blocks']} blocks learned after the question"
184
+ + (f", {r['missing_blocks']} evidence ids not in the bundle" if r["missing_blocks"] else ""))
185
+ sys.exit(1 if r["leaked_questions"] else 0)
186
+ elif a.cmd == "validate":
187
+ checks = validate(a.bundle)
188
+ for c in checks:
189
+ print(f"{'ok ' if c.ok else 'FAIL'} {c.check_id:<20} {c.detail}")
190
+ sys.exit(0 if all(c.ok for c in checks) else 1)
191
+ elif a.cmd == "sync":
192
+ token = a.token or os.environ.get("TCKG_TOKEN") or p.error("--token or $TCKG_TOKEN is required")
193
+ r = sync(a.bundle, TckgStore(a.url, token, a.space), a.as_of, push=not a.pull_only, pull=not a.push_only)
194
+ pu, pl = r["pushed"], r["pulled"]
195
+ parts = []
196
+ if pu is not None:
197
+ parts.append(f"pushed {pu['accepted']} rows in {pu['batches']} batches" + (f", {len(pu['skipped'])} already there" if pu["skipped"] else "")
198
+ + (f", {len(pu['refused'])} refused" if pu["refused"] else "") + (f", {pu['resolutions_not_pushed']} resolutions kept local" if pu["resolutions_not_pushed"] else ""))
199
+ if pl is not None:
200
+ parts.append(f"pulled {pl['nodes']} nodes, {pl['edges']} edges, {pl['resolutions']} resolutions")
201
+ print(f"{a.bundle} <-> {a.url}: " + "; ".join(parts) + f" (as of {r['as_of'][:19]})")
202
+ for x in (pu or {}).get("refused", []):
203
+ print(f" refused {x}")
204
+ sys.exit(1 if pu and pu["refused"] else 0)
205
+ elif a.cmd == "to-claimreview":
206
+ out_json(to_claimreview(a.bundle, a.as_of, a.base_url, a.valid_at))
207
+ elif a.cmd == "to-okf":
208
+ out = to_okf(a.bundle, a.as_of, a.out, a.valid_at)
209
+ print(f"wrote {out}: {len(list(out.glob('*.md')))} documents")
210
+ elif a.cmd == "to-parquet":
211
+ out = write_parquet(a.bundle, a.out)
212
+ print(f"wrote {out}: {sorted(x.name for x in out.iterdir())}")
213
+ elif a.cmd == "from-factcheck":
214
+ if a.from_json:
215
+ d = json.load(open(a.from_json))
216
+ claims = d.get("claims", []) if isinstance(d, dict) else d
217
+ elif a.query:
218
+ key = a.api_key or os.environ.get("FACTCHECK_API_KEY") or p.error("--api-key or $FACTCHECK_API_KEY is required with --query")
219
+ claims = factcheck_search(a.query, key, a.language, a.pages)
220
+ else:
221
+ p.error("one of --query or --from-json is required")
222
+ existing = Bundle(a.out) if (pathlib.Path(a.out) / "factblock.json").exists() else None
223
+ r = bundle_from_factcheck(claims, existing, a.namespace)
224
+ write_bundle(r, a.out, append=True)
225
+ s_ = r["summary"]
226
+ print(f"{a.out}: +{s_['claims']} claims, +{s_['verdicts']} verdicts, {s_['batches']} batches")
227
+
228
+
229
+ def run():
230
+ try:
231
+ main()
232
+ except BrokenPipeError: # `factblock scan ... | head`
233
+ sys.stderr.close()
234
+
235
+
236
+ if __name__ == "__main__":
237
+ run()
@@ -0,0 +1,2 @@
1
+ """Writers that turn another store's data into a FactBlock bundle (SPEC 6: a writer that
2
+ is not a ledger declares backfill batches; it never invents known_at)."""
@@ -0,0 +1,106 @@
1
+ """Fact-check feeds -> FactBlock bundle (SPEC 9.2, reverse direction). Reads the shape of Google's
2
+ Fact Check Tools API (claims:search): a claim with `text`, `claimant`, `claimDate` and its
3
+ `claimReview[]` with `publisher`, `url`, `title`, `reviewDate`, `textualRating`. Each claim becomes a
4
+ node, each review a resolution row. The feed's dates are self-reported, so every row is written
5
+ under a backfill batch declared at its review date (section 6): the claim became known to the
6
+ fact-checking world when it was first reviewed, and each verdict when it was published.
7
+
8
+ bundle_from_factcheck(claims, existing=None) -> {"manifest", "nodes", "edges", "resolutions"}
9
+ search(query, api_key, language=None, pages=1) -> claims # the live API; needs a Google API key
10
+ """
11
+ import hashlib
12
+ import json
13
+ import urllib.error
14
+ import urllib.parse
15
+ import urllib.request
16
+ from datetime import datetime, timezone
17
+
18
+ from ..bundle import parse_instant
19
+
20
+ LEDGER = "factblock-factcheck-adapter/0.1"
21
+ API = "https://factchecktools.googleapis.com/v1alpha1/claims:search"
22
+ # textualRating, lowercased and stripped of punctuation -> recommended outcome (SPEC 3.6). Unknown ratings keep value only.
23
+ OUTCOMES = {"true": "true", "correct": "true", "accurate": "true", "mostly true": "mostly_true", "mostly correct": "mostly_true",
24
+ "partly true": "mostly_true", "partially true": "mostly_true", "half true": "misleading", "misleading": "misleading",
25
+ "missing context": "misleading", "mostly false": "mostly_false", "partly false": "mostly_false", "false": "false",
26
+ "incorrect": "false", "pants on fire": "false", "fake": "false", "unproven": "unverifiable", "unverifiable": "unverifiable",
27
+ "unsupported": "unverifiable", "no evidence": "unverifiable"}
28
+
29
+
30
+ def _iso(v):
31
+ d = parse_instant(v)
32
+ return d.astimezone(timezone.utc).isoformat()
33
+
34
+
35
+ def _outcome(rating):
36
+ key = "".join(c if c.isalnum() or c == " " else " " for c in (rating or "").lower()).split()
37
+ return OUTCOMES.get(" ".join(key))
38
+
39
+
40
+ def search(query, api_key, language=None, pages=1, page_size=100):
41
+ """claims:search, following nextPageToken up to `pages` times. Returns the raw claim dicts."""
42
+ out, token = [], None
43
+ for _ in range(pages):
44
+ q = {"query": query, "key": api_key, "pageSize": page_size, **({"languageCode": language} if language else {}), **({"pageToken": token} if token else {})}
45
+ try:
46
+ with urllib.request.urlopen(API + "?" + urllib.parse.urlencode(q), timeout=60) as resp:
47
+ d = json.load(resp)
48
+ except urllib.error.HTTPError as e: # Google puts the reason in the body; show it instead of a bare status
49
+ body = e.read().decode(errors="replace")
50
+ try:
51
+ body = json.loads(body)["error"]["message"]
52
+ except (ValueError, KeyError, TypeError):
53
+ pass
54
+ raise RuntimeError(f"Fact Check Tools API: HTTP {e.code}: {body}") from None
55
+ out += d.get("claims", [])
56
+ token = d.get("nextPageToken")
57
+ if not token:
58
+ break
59
+ return out
60
+
61
+
62
+ def bundle_from_factcheck(claims, existing=None, namespace="factcheck", captured_at=None) -> dict:
63
+ captured_at = _iso(captured_at or datetime.now(timezone.utc))
64
+ have_nodes = {n["id"] for n in (existing.nodes if existing else [])}
65
+ have_res = {(r["target_id"], r["decided_at"].isoformat()) for r in (existing.resolutions if existing else [])}
66
+ batches, nodes, resolutions = {}, [], []
67
+
68
+ def stamp(at):
69
+ batches.setdefault(at, f"factcheck-{at[:10]}-{hashlib.sha1(at.encode()).hexdigest()[:6]}")
70
+ return {"known_at": at, "attestation": {"ledger": LEDGER, "batch": batches[at]}}
71
+
72
+ for c in claims:
73
+ reviews = [r for r in c.get("claimReview", []) if r.get("reviewDate")]
74
+ if not c.get("text") or not reviews:
75
+ continue
76
+ reviews.sort(key=lambda r: r["reviewDate"])
77
+ first = _iso(reviews[0]["reviewDate"])
78
+ asserted = _iso(c["claimDate"]) if c.get("claimDate") else first
79
+ nid = hashlib.sha1(f"{c['text']}|{c.get('claimant', '')}|{asserted}".encode()).hexdigest()[:16]
80
+ if nid not in have_nodes:
81
+ have_nodes.add(nid)
82
+ payload = {"language": reviews[0].get("languageCode")}
83
+ if c.get("claimant"):
84
+ payload["speaker"] = c["claimant"]
85
+ nodes.append({"id": nid, "kind": "claim", "statement": c["text"], "payload": {k: v for k, v in payload.items() if v},
86
+ "asserted_at": asserted, "valid_from": asserted, "valid_to": None, **stamp(first)})
87
+ for r in reviews:
88
+ decided = _iso(r["reviewDate"])
89
+ if (nid, decided) in have_res:
90
+ continue
91
+ have_res.add((nid, decided))
92
+ pub = r.get("publisher") or {}
93
+ row = {"target_id": nid, "value": r.get("textualRating"), "method": "claimreview", "decided_at": decided, **stamp(decided),
94
+ "evidence": [{"type": "FACT_CHECK", "url": r.get("url"), "title": r.get("title"), "publisher": pub.get("name"), "published_at": decided}]}
95
+ if pub.get("site"):
96
+ row["resolver"] = f"org:{pub['site']}"
97
+ if _outcome(r.get("textualRating")):
98
+ row["outcome"] = _outcome(r["textualRating"])
99
+ resolutions.append(row)
100
+ manifest = {"factblock_version": "1.0-draft.1", "namespace": namespace, "source": LEDGER, "exported_as_of": captured_at,
101
+ "tables": {"nodes": "nodes.jsonl", "edges": "edges.jsonl", "resolutions": "resolutions.jsonl"},
102
+ "declarations": {"facts": [], "edge_types": [], "embedding": None,
103
+ "backfills": [{"batch": name, "declared_known_at": at, "reason": "fact-check feed: the publisher's review date is self-reported",
104
+ "declared_by": "process:" + LEDGER, "captured_at": captured_at} for at, name in sorted(batches.items())]}}
105
+ return {"manifest": manifest, "nodes": nodes, "edges": [], "resolutions": resolutions,
106
+ "summary": {"claims": len(nodes), "verdicts": len(resolutions), "batches": len(batches)}}
@@ -0,0 +1,110 @@
1
+ """Graphiti (Zep) graph -> FactBlock bundle. The second implementation of the format:
2
+ it reads Graphiti's own objects, not tckg, and depends on nothing but the duck-typed
3
+ attributes below, so it runs against any Graphiti backend (Neo4j, FalkorDB, Kuzu).
4
+
5
+ Mapping (SPEC.md 3; tckg's memory/tckg_driver is the reverse direction):
6
+ group_id -> space
7
+ EpisodicNode -> node kind=episode (statement=name, asserted_at=valid_at)
8
+ EntityNode -> node kind=entity (statement=name)
9
+ EntityEdge (a fact) -> node kind=claim (statement=fact, valid=[valid_at, invalid_at))
10
+ + claim MENTIONS source entity, claim MENTIONS target entity
11
+ + claim DERIVED_FROM each episode it came from
12
+ EpisodicEdge -> edge episode MENTIONS entity
13
+ created_at -> known_at, attested by one backfill batch per distinct created_at:
14
+ Graphiti stamps created_at itself, so it is self-reported (SPEC 6, 9)
15
+ Embeddings are not exported: the bundle cannot name the model (declarations.embedding is null).
16
+ """
17
+ import json
18
+ from datetime import datetime, timezone
19
+ from pathlib import Path
20
+
21
+ LEDGER = "factblock-graphiti-adapter/0.1"
22
+
23
+
24
+ def _iso(d):
25
+ if d is None:
26
+ return None
27
+ if d.tzinfo is None:
28
+ d = d.replace(tzinfo=timezone.utc)
29
+ return d.astimezone(timezone.utc).isoformat()
30
+
31
+
32
+ def _enum(v):
33
+ return getattr(v, "value", v)
34
+
35
+
36
+ def bundle_from_graphiti(entities, episodes, facts, episodic_edges, exported_at=None) -> dict:
37
+ """Returns {"manifest", "nodes", "edges"} as plain dicts; write_bundle() puts them on disk."""
38
+ exported_at = _iso(exported_at or datetime.now(timezone.utc))
39
+ # Batches are numbered in time order, whatever order the store returned rows in,
40
+ # so an importer that applies them by number applies them chronologically and
41
+ # never sees an edge before its endpoints.
42
+ instants = sorted({_iso(x.created_at) for group in (entities, episodes, facts, episodic_edges) for x in group})
43
+ batches = {k: f"graphiti-{i + 1}" for i, k in enumerate(instants)}
44
+
45
+ def batch_for(created_at):
46
+ return batches[_iso(created_at)]
47
+
48
+ def stamp(created_at):
49
+ return {"known_at": _iso(created_at), "attestation": {"ledger": LEDGER, "batch": batch_for(created_at)}}
50
+
51
+ nodes, edges, spaces = [], [], set()
52
+ ids = set()
53
+ for e in episodes:
54
+ spaces.add(e.group_id); ids.add(e.uuid)
55
+ nodes.append({"id": e.uuid, "kind": "episode", "space": e.group_id, "statement": e.name,
56
+ "payload": {"source": _enum(e.source), "source_description": e.source_description,
57
+ "content": e.content, "labels": list(getattr(e, "labels", []) or [])},
58
+ "asserted_at": _iso(e.valid_at), "valid_from": _iso(e.valid_at), "valid_to": None, **stamp(e.created_at)})
59
+ for n in entities:
60
+ spaces.add(n.group_id); ids.add(n.uuid)
61
+ nodes.append({"id": n.uuid, "kind": "entity", "space": n.group_id, "statement": n.name,
62
+ "payload": {"labels": list(getattr(n, "labels", []) or []), "summary": getattr(n, "summary", "") or "",
63
+ "attributes": dict(getattr(n, "attributes", {}) or {})},
64
+ "asserted_at": _iso(n.created_at), "valid_from": _iso(n.created_at), "valid_to": None, **stamp(n.created_at)})
65
+ seen_edges = set()
66
+
67
+ def edge(src, dst, typ, when, created_at):
68
+ key = (src, dst, typ)
69
+ if key in seen_edges or src not in ids or dst not in ids:
70
+ return
71
+ seen_edges.add(key)
72
+ edges.append({"source_id": src, "target_id": dst, "edge_type": typ,
73
+ "asserted_at": _iso(when), "valid_from": _iso(when), "valid_to": None, **stamp(created_at)})
74
+
75
+ for f in facts:
76
+ spaces.add(f.group_id); ids.add(f.uuid)
77
+ start = f.valid_at or f.created_at
78
+ end = f.invalid_at if (f.invalid_at and start and f.invalid_at > start) else None
79
+ payload = {"name": f.name, "episodes": list(f.episodes or []), "attributes": dict(getattr(f, "attributes", {}) or {})}
80
+ if getattr(f, "expired_at", None):
81
+ payload["graphiti_expired_at"] = _iso(f.expired_at)
82
+ if f.invalid_at and end is None:
83
+ payload["graphiti_invalid_at"] = _iso(f.invalid_at) # kept, not applied: it did not follow valid_at
84
+ nodes.append({"id": f.uuid, "kind": "claim", "space": f.group_id, "statement": f.fact, "payload": payload,
85
+ "asserted_at": _iso(start), "valid_from": _iso(start), "valid_to": _iso(end), **stamp(f.created_at)})
86
+ for f in facts:
87
+ start = f.valid_at or f.created_at
88
+ edge(f.uuid, f.source_node_uuid, "MENTIONS", start, f.created_at)
89
+ edge(f.uuid, f.target_node_uuid, "MENTIONS", start, f.created_at)
90
+ for ep in f.episodes or []:
91
+ edge(f.uuid, ep, "DERIVED_FROM", start, f.created_at)
92
+ for x in episodic_edges:
93
+ edge(x.source_node_uuid, x.target_node_uuid, "MENTIONS", x.created_at, x.created_at)
94
+
95
+ manifest = {
96
+ "factblock_version": "1.0-draft.1",
97
+ "namespace": "graphiti/" + ",".join(sorted(spaces)),
98
+ "source": LEDGER,
99
+ "exported_as_of": exported_at,
100
+ "tables": {"nodes": "nodes.jsonl", "edges": "edges.jsonl"},
101
+ "declarations": {
102
+ "facts": [],
103
+ "backfills": [{"batch": b, "declared_known_at": k, "reason": "Graphiti created_at, self-reported by the source store",
104
+ "declared_by": "process:" + LEDGER, "captured_at": exported_at} for k, b in batches.items()],
105
+ "edge_types": [], "embedding": None},
106
+ }
107
+ return {"manifest": manifest, "nodes": nodes, "edges": edges}
108
+
109
+
110
+ from ..bundle import write_bundle # noqa: E402,F401 (kept here for callers that imported it from the adapter)
factblock/bundle.py ADDED
@@ -0,0 +1,89 @@
1
+ """Load a bundle (SPEC 5.1): manifest plus node, edge, resolution tables as lists of dicts."""
2
+ import json
3
+ from datetime import datetime, timezone
4
+ from pathlib import Path
5
+
6
+ INSTANT_KEYS = ("asserted_at", "valid_from", "valid_to", "known_at", "decided_at")
7
+
8
+
9
+ def parse_instant(v):
10
+ """RFC 3339 string -> aware datetime. None stays None."""
11
+ if v is None or isinstance(v, datetime):
12
+ return v
13
+ d = datetime.fromisoformat(v.replace("Z", "+00:00"))
14
+ return d if d.tzinfo else d.replace(tzinfo=timezone.utc)
15
+
16
+
17
+ def _read_table(path: Path) -> list[dict]:
18
+ if not path.exists():
19
+ return []
20
+ if path.suffix == ".parquet":
21
+ import pyarrow.parquet as pq
22
+ from .parquet import decode_parquet_row
23
+ rows = [decode_parquet_row(r) for r in pq.read_table(path).to_pylist()]
24
+ else:
25
+ with path.open() as f:
26
+ rows = [json.loads(line) for line in f if line.strip()]
27
+ for r in rows:
28
+ for k in INSTANT_KEYS:
29
+ if k in r:
30
+ r[k] = parse_instant(r[k])
31
+ return rows
32
+
33
+
34
+ class Bundle:
35
+ def __init__(self, path):
36
+ self.path = Path(path)
37
+ self.manifest = json.loads((self.path / "factblock.json").read_text())
38
+ tables = self.manifest.get("tables", {})
39
+ self.nodes = _read_table(self.path / tables.get("nodes", "nodes.jsonl"))
40
+ self.edges = _read_table(self.path / tables.get("edges", "edges.jsonl"))
41
+ self.resolutions = _read_table(self.path / tables.get("resolutions", "resolutions.jsonl"))
42
+ d = self.manifest.get("declarations", {})
43
+ self.facts = {}
44
+ for f in d.get("facts", []):
45
+ f["declared_at"] = parse_instant(f.get("declared_at"))
46
+ self.facts.setdefault(f["fact_key"], []).append(f)
47
+ self.backfills = {b["batch"]: b for b in d.get("backfills", [])}
48
+ for b in self.backfills.values():
49
+ b["declared_known_at"] = parse_instant(b["declared_known_at"])
50
+ self.edge_types = {e["edge_type"]: e["family"] for e in d.get("edge_types", [])}
51
+
52
+ def blocks(self):
53
+ """Every block with a table label: nodes, edges, resolutions."""
54
+ for r in self.nodes:
55
+ yield "node", r
56
+ for r in self.edges:
57
+ yield "edge", r
58
+ for r in self.resolutions:
59
+ yield "resolution", r
60
+
61
+
62
+ def write_bundle(result: dict, out, append: bool = False) -> Path:
63
+ """Put {"manifest", "nodes", "edges", "resolutions"?} on disk as a JSONL bundle. With append=True and a bundle
64
+ already at `out`, rows are added and the manifest's declarations are merged, so a folder grows
65
+ one batch at a time and stays valid."""
66
+ out = Path(out); out.mkdir(parents=True, exist_ok=True)
67
+ mpath = out / "factblock.json"
68
+ manifest = result["manifest"]
69
+ if append and mpath.exists():
70
+ old = json.loads(mpath.read_text())
71
+ for key in ("facts", "backfills", "edge_types"):
72
+ have = {json.dumps(x, sort_keys=True) for x in old["declarations"].get(key, [])}
73
+ old["declarations"][key] = old["declarations"].get(key, []) + [x for x in manifest["declarations"].get(key, []) if json.dumps(x, sort_keys=True) not in have]
74
+ old["exported_as_of"] = manifest.get("exported_as_of", old.get("exported_as_of"))
75
+ manifest = old
76
+ mode = "a"
77
+ else:
78
+ mode = "w"
79
+ tables = manifest.setdefault("tables", {})
80
+ tables.setdefault("nodes", "nodes.jsonl"); tables.setdefault("edges", "edges.jsonl")
81
+ if result.get("resolutions"):
82
+ tables.setdefault("resolutions", "resolutions.jsonl")
83
+ mpath.write_text(json.dumps(manifest, indent=1) + "\n")
84
+ for name in ("nodes", "edges", "resolutions"):
85
+ rows = result.get(name, [])
86
+ if name in tables and (rows or (mode == "w" and name != "resolutions")):
87
+ with (out / tables[name]).open(mode) as f:
88
+ f.write("".join(json.dumps({k: v for k, v in r.items() if not k.startswith("_")}, default=lambda d: d.isoformat()) + "\n" for r in rows))
89
+ return out
@@ -0,0 +1,64 @@
1
+ """SPEC 9: the ClaimReview projection. Every verdict visible as of an instant becomes one schema.org
2
+ ClaimReview (JSON-LD), the shape Google's fact-check tools, Meta's programme and ClaimReview-aware
3
+ CMSs consume. Lossy by design: one verdict per block (the latest), no chain, no interval. What
4
+ ClaimReview cannot say travels under `factblock:` keys that a consumer may ignore."""
5
+ from .bundle import Bundle, parse_instant
6
+ from .scan import _as_of, _visible, scan
7
+
8
+ # outcome -> (alternateName, ratingValue on a 1..5 scale, or None when the scale does not apply)
9
+ RATINGS = {
10
+ "true": ("True", 5), "mostly_true": ("Mostly true", 4), "mostly_false": ("Mostly false", 2), "false": ("False", 1),
11
+ "misleading": ("Misleading", 2), "unverifiable": ("Unverifiable", None),
12
+ "came_true": ("Came true", 5), "partial": ("Partially came true", 3), "did_not": ("Did not come true", 1), "undecidable": ("Undecidable", None),
13
+ }
14
+
15
+
16
+ def _actor(a):
17
+ if not a:
18
+ return None
19
+ name = a.split(":", 1)[1] if a.startswith(("human:", "process:", "org:")) else a.split("/", 1)[0]
20
+ return {"@type": "Person" if a.startswith("human:") else "Organization", "name": name}
21
+
22
+
23
+ def _rating(r):
24
+ outcome = r.get("outcome") or (r["value"] if isinstance(r.get("value"), str) else None)
25
+ name, value = RATINGS.get(outcome, (str(outcome) if outcome is not None else str(r.get("value")), None))
26
+ out = {"@type": "Rating", "alternateName": name}
27
+ if value is not None:
28
+ out.update({"ratingValue": value, "bestRating": 5, "worstRating": 1})
29
+ return out
30
+
31
+
32
+ def to_claimreview(bundle, as_of, base_url=None, valid_at=None) -> dict:
33
+ b = bundle if isinstance(bundle, Bundle) else Bundle(bundle)
34
+ s = scan(b, as_of, valid_at)
35
+ t = _as_of(as_of); v = parse_instant(valid_at) if valid_at else t
36
+ nodes = {n["id"]: n for n in b.nodes}
37
+ latest = {}
38
+ for r in sorted((r for r in b.resolutions if _visible(r, t, v)), key=lambda r: (r["decided_at"], r["known_at"])):
39
+ latest[r["target_id"]] = r # ClaimReview has no history: the latest verdict stands for the block
40
+ reviews = []
41
+ for tid, r in latest.items():
42
+ n = nodes.get(tid)
43
+ if n is None:
44
+ continue
45
+ payload = n.get("payload") or {}
46
+ source = payload.get("source") if isinstance(payload.get("source"), dict) else {}
47
+ claim = {"@type": "Claim", "datePublished": n["asserted_at"].date().isoformat()}
48
+ if payload.get("speaker"):
49
+ claim["author"] = {"@type": "Person", "name": payload["speaker"]}
50
+ elif n.get("author"):
51
+ claim["author"] = _actor(n["author"])
52
+ if source.get("url"):
53
+ claim["appearance"] = {"@type": "CreativeWork", "url": source["url"], **({"name": source["title"]} if source.get("title") else {})}
54
+ cr = {"@type": "ClaimReview", "claimReviewed": n.get("statement"), "datePublished": r["decided_at"].date().isoformat(),
55
+ "itemReviewed": claim, "reviewRating": _rating(r), "factblock:id": tid, "factblock:kind": n["kind"],
56
+ "factblock:known_at": r["known_at"].isoformat()}
57
+ if _actor(r.get("resolver")):
58
+ cr["author"] = _actor(r["resolver"])
59
+ if base_url:
60
+ cr["url"] = f"{base_url.rstrip('/')}/{tid}"
61
+ if r.get("method"):
62
+ cr["factblock:method"] = r["method"]
63
+ reviews.append(cr)
64
+ return {"@context": "https://schema.org", "@graph": reviews, "factblock:certificate": s.certificate}