@tangle-network/agent-knowledge 6.1.11 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +9 -0
- package/CHANGELOG.md +25 -0
- package/README.md +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CmW6iORW.js → benchmarks-C8L7HJb4.js} +2 -2
- package/dist/{benchmarks-CmW6iORW.js.map → benchmarks-C8L7HJb4.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{ids-DRqPZ42_.js → ids-Bevz_pXV.js} +6 -2
- package/dist/ids-Bevz_pXV.js.map +1 -0
- package/dist/{index-CIW3G4s_.d.ts → index-D0wc4GYg.d.ts} +2 -2
- package/dist/{index-CIW3G4s_.d.ts.map → index-D0wc4GYg.d.ts.map} +1 -1
- package/dist/{index-CGBctbit.d.ts → index-Dwp3Mx-w.d.ts} +3 -3
- package/dist/{index-CGBctbit.d.ts.map → index-Dwp3Mx-w.d.ts.map} +1 -1
- package/dist/index.d.ts +377 -43
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +294 -198
- package/dist/index.js.map +1 -1
- package/dist/{inspect-DAXpFyrs.js → inspect-CJQGYuKa.js} +809 -17
- package/dist/inspect-CJQGYuKa.js.map +1 -0
- package/dist/memory/index.d.ts +2 -2
- package/dist/memory/index.js +2 -2
- package/dist/{memory-C6KPRhoU.js → memory-CIYRB_Q8.js} +3 -3
- package/dist/{memory-C6KPRhoU.js.map → memory-CIYRB_Q8.js.map} +1 -1
- package/dist/sources/index.js +1 -1
- package/dist/types-BOfmvDe-.d.ts +306 -0
- package/dist/{types-DcCCzreS.d.ts.map → types-BOfmvDe-.d.ts.map} +1 -1
- package/dist/viz/index.d.ts +1 -1
- package/docs/architecture.md +32 -0
- package/package.json +1 -1
- package/dist/ids-DRqPZ42_.js.map +0 -1
- package/dist/inspect-DAXpFyrs.js.map +0 -1
- package/dist/types-DcCCzreS.d.ts +0 -175
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { n as slugify, r as stableId, t as sha256 } from "./ids-
|
|
1
|
+
import { i as textSourceId, n as slugify, r as stableId, t as sha256 } from "./ids-Bevz_pXV.js";
|
|
2
2
|
import { n as searchKnowledge } from "./search-CP0QtBJZ.js";
|
|
3
3
|
import { createHash, randomUUID } from "node:crypto";
|
|
4
4
|
import { lstat, mkdir, mkdtemp, open, readFile, readdir, realpath, rename, rm } from "node:fs/promises";
|
|
@@ -1208,6 +1208,413 @@ function addSourceOverlapEdges(pages, edges) {
|
|
|
1208
1208
|
}
|
|
1209
1209
|
}
|
|
1210
1210
|
//#endregion
|
|
1211
|
+
//#region src/claim-ledger.ts
|
|
1212
|
+
/**
|
|
1213
|
+
* The algebra of a research claim ledger: claim identity, and how two ledgers
|
|
1214
|
+
* that accumulated evidence for the same goal combine into one.
|
|
1215
|
+
*
|
|
1216
|
+
* This lives apart from `research-driving-driver.ts` because it is no longer
|
|
1217
|
+
* that driver's private business. A ledger is a durable record now, and a
|
|
1218
|
+
* durable record addressed by id is a record two writers can reach: two rounds
|
|
1219
|
+
* of one run resuming from disk, or two workers researching one goal in
|
|
1220
|
+
* parallel. `putClaimLedger` writes the whole record, so the second writer's
|
|
1221
|
+
* write erases the first writer's claims — the ledger persists and the
|
|
1222
|
+
* knowledge still does not compound. Combining is therefore part of what the
|
|
1223
|
+
* record MEANS, and it belongs next to the record rather than inside one
|
|
1224
|
+
* consumer of it.
|
|
1225
|
+
*
|
|
1226
|
+
* Every combination here is monotone: support only grows, contradiction edges
|
|
1227
|
+
* only accumulate, `contested` only latches on, `firstSeenRound` only moves
|
|
1228
|
+
* earlier. That is what makes it safe to apply twice — a retried write cannot
|
|
1229
|
+
* produce a different ledger than a single write did.
|
|
1230
|
+
*/
|
|
1231
|
+
/**
|
|
1232
|
+
* Claim identity = sha256 of the normalized claim text, so the same assertion
|
|
1233
|
+
* discovered independently by two workers is ONE claim with two supporting
|
|
1234
|
+
* sources rather than two claims with one each — which is the difference
|
|
1235
|
+
* between corroborated and unsupported.
|
|
1236
|
+
*/
|
|
1237
|
+
function claimId(text) {
|
|
1238
|
+
return `c_${sha256(normalizeClaimText(text)).slice(0, 16)}`;
|
|
1239
|
+
}
|
|
1240
|
+
/** Stable identity for one extracted claim/source/contradiction observation. */
|
|
1241
|
+
function claimEvidenceId(claim) {
|
|
1242
|
+
return `e_${sha256(JSON.stringify([
|
|
1243
|
+
claim.claimId,
|
|
1244
|
+
claim.sourceId,
|
|
1245
|
+
claim.sourceUri,
|
|
1246
|
+
claim.sourceContentHash,
|
|
1247
|
+
claim.contradictsClaimId ?? null
|
|
1248
|
+
])).slice(0, 16)}`;
|
|
1249
|
+
}
|
|
1250
|
+
/** Canonical key for one exact registry-id + original-URI + hash source version. */
|
|
1251
|
+
function researchSourceVersionKey(source) {
|
|
1252
|
+
return JSON.stringify([
|
|
1253
|
+
source.sourceId,
|
|
1254
|
+
source.uri,
|
|
1255
|
+
source.contentHash
|
|
1256
|
+
]);
|
|
1257
|
+
}
|
|
1258
|
+
/** Case-, whitespace-, and stylistic-punctuation-insensitive claim identity form. */
|
|
1259
|
+
function normalizeClaimText(text) {
|
|
1260
|
+
return text.normalize("NFKC").toLowerCase().replace(/<=|≤/gu, " symbol_less_than_or_equal ").replace(/>=|≥/gu, " symbol_greater_than_or_equal ").replace(/!=|≠/gu, " symbol_not_equal ").replace(/==|=/gu, " symbol_equal ").replace(/±/gu, " symbol_plus_or_minus ").replace(/[≈~]/gu, " symbol_approximately ").replace(/<|←|⇐/gu, " symbol_less_or_left ").replace(/>|→|⇒/gu, " symbol_greater_or_right ").replace(/[+]/gu, " symbol_plus_or_positive ").replace(/[-−]/gu, " symbol_minus_or_negative ").replace(/[^\p{L}\p{N}\s]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
1261
|
+
}
|
|
1262
|
+
/**
|
|
1263
|
+
* The canonical host a source uri counts as, which is what makes two sources
|
|
1264
|
+
* INDEPENDENT: corroboration is "distinct hosts", so this function is the rule
|
|
1265
|
+
* that decides whether a claim is confirmed or merely repeated. Exported so a
|
|
1266
|
+
* consumer building a `ResearchClaimRecord` cannot answer it a different way — a
|
|
1267
|
+
* consumer that counted raw uris would report two pages of one site as
|
|
1268
|
+
* independent confirmation.
|
|
1269
|
+
*/
|
|
1270
|
+
function claimSourceHost(uri) {
|
|
1271
|
+
try {
|
|
1272
|
+
return new URL(uri.trim()).hostname.toLowerCase().replace(/^www\./, "");
|
|
1273
|
+
} catch {
|
|
1274
|
+
return uri.trim().toLowerCase();
|
|
1275
|
+
}
|
|
1276
|
+
}
|
|
1277
|
+
/** Stable identity for one deep question. */
|
|
1278
|
+
function deepQuestionId(kind, text) {
|
|
1279
|
+
return `q_${sha256(`${kind}:${text}`).slice(0, 16)}`;
|
|
1280
|
+
}
|
|
1281
|
+
/**
|
|
1282
|
+
* Refuse a claim record whose identity or source count disagrees with its evidence.
|
|
1283
|
+
*
|
|
1284
|
+
* `supportingHosts` is used as the independent-source count, so accepting hosts
|
|
1285
|
+
* that cannot be derived from `supportingUris` would let a malformed record
|
|
1286
|
+
* manufacture corroboration. Canonical ordering also makes equal records have
|
|
1287
|
+
* equal bytes regardless of which process assembled them.
|
|
1288
|
+
*/
|
|
1289
|
+
function assertTrackedClaimIntegrity(claim) {
|
|
1290
|
+
if (claim.id !== claimId(claim.text)) throw new Error(`claim '${claim.id}' does not match its text-derived identity`);
|
|
1291
|
+
if (claim.text !== claim.text.trim()) throw new Error(`claim '${claim.id}' text must not have surrounding whitespace`);
|
|
1292
|
+
if (claim.supportingUris.length === 0) throw new Error(`claim '${claim.id}' must have registered supporting evidence`);
|
|
1293
|
+
assertSortedUnique(`claim '${claim.id}' supportingUris`, claim.supportingUris);
|
|
1294
|
+
assertSortedUnique(`claim '${claim.id}' supportingHosts`, claim.supportingHosts);
|
|
1295
|
+
assertSortedUnique(`claim '${claim.id}' contradicts`, claim.contradicts);
|
|
1296
|
+
const expectedHosts = [...new Set(claim.supportingUris.map(claimSourceHost).filter(Boolean))].sort();
|
|
1297
|
+
if (!sameStrings(claim.supportingHosts, expectedHosts)) throw new Error(`claim '${claim.id}' supportingHosts must equal the hosts derived from supportingUris`);
|
|
1298
|
+
if (claim.contradicts.includes(claim.id)) throw new Error(`claim '${claim.id}' cannot contradict itself`);
|
|
1299
|
+
if (claim.contradicts.length > 0 && !claim.contested) throw new Error(`claim '${claim.id}' with a contradiction must be contested`);
|
|
1300
|
+
if (claim.contested && claim.contradicts.length === 0) throw new Error(`claim '${claim.id}' cannot be contested without a contradiction`);
|
|
1301
|
+
}
|
|
1302
|
+
/** Refuse a deep question whose stable identity or set fields are malformed. */
|
|
1303
|
+
function assertDeepQuestionIntegrity(question) {
|
|
1304
|
+
if (question.id !== deepQuestionId(question.kind, question.text)) throw new Error(`question '${question.id}' does not match its kind-and-text identity`);
|
|
1305
|
+
if (question.text !== question.text.trim()) throw new Error(`question '${question.id}' text must not have surrounding whitespace`);
|
|
1306
|
+
assertSortedUnique(`question '${question.id}' claimIds`, question.claimIds);
|
|
1307
|
+
}
|
|
1308
|
+
/** Refuse an extracted observation whose identity or content is malformed. */
|
|
1309
|
+
function assertResearchClaimEvidenceIntegrity(evidence) {
|
|
1310
|
+
if (evidence.claimId !== claimId(evidence.text)) throw new Error(`claim evidence '${evidence.id}' does not match its text-derived claim identity`);
|
|
1311
|
+
if (evidence.id !== claimEvidenceId(evidence)) throw new Error(`claim evidence '${evidence.id}' does not match its content-derived identity`);
|
|
1312
|
+
if (evidence.text !== evidence.text.trim()) throw new Error(`claim evidence '${evidence.id}' text must not have surrounding whitespace`);
|
|
1313
|
+
if (!/^[a-f0-9]{64}$/.test(evidence.sourceContentHash)) throw new Error(`claim evidence '${evidence.id}' sourceContentHash must be a SHA-256 digest`);
|
|
1314
|
+
const expectedSourceId = textSourceId(evidence.sourceUri, evidence.sourceContentHash);
|
|
1315
|
+
if (evidence.sourceId !== expectedSourceId) throw new Error(`claim evidence '${evidence.id}' sourceId does not match URI-and-content identity`);
|
|
1316
|
+
if (evidence.contradictsClaimId === evidence.claimId) throw new Error(`claim evidence '${evidence.id}' cannot contradict its own claim`);
|
|
1317
|
+
}
|
|
1318
|
+
/** Refuse a ledger that is not one canonical, internally consistent record. */
|
|
1319
|
+
function assertResearchClaimLedgerIntegrity(ledger) {
|
|
1320
|
+
if (ledger.schemaVersion !== 2) throw new Error(`claim ledger '${ledger.id}' must use schema version 2`);
|
|
1321
|
+
if (ledger.preparedRounds !== void 0 && ledger.preparedRounds <= ledger.rounds) throw new Error(`claim ledger '${ledger.id}' preparedRounds must be greater than completed rounds`);
|
|
1322
|
+
assertSortedUnique(`claim ledger '${ledger.id}' claimEvidence`, ledger.claimEvidence.map((evidence) => evidence.id));
|
|
1323
|
+
for (const source of ledger.registeredSources) {
|
|
1324
|
+
if (source.sourceId.length === 0 || source.uri.length === 0 || !/^[a-f0-9]{64}$/.test(source.contentHash)) throw new Error(`claim ledger '${ledger.id}' contains an invalid registered source version`);
|
|
1325
|
+
const expectedSourceId = textSourceId(source.uri, source.contentHash);
|
|
1326
|
+
if (source.sourceId !== expectedSourceId) throw new Error(`registered source '${source.sourceId}' does not match URI-and-content identity '${expectedSourceId}'`);
|
|
1327
|
+
}
|
|
1328
|
+
assertSortedUnique(`claim ledger '${ledger.id}' registeredSources`, ledger.registeredSources.map((source) => source.sourceId));
|
|
1329
|
+
assertSortedUnique(`claim ledger '${ledger.id}' claims`, ledger.claims.map((claim) => claim.id));
|
|
1330
|
+
assertSortedUnique(`claim ledger '${ledger.id}' questions`, ledger.questions.map((question) => question.id));
|
|
1331
|
+
for (const evidence of ledger.claimEvidence) assertResearchClaimEvidenceIntegrity(evidence);
|
|
1332
|
+
const registeredSources = new Set(ledger.registeredSources.map(researchSourceVersionKey));
|
|
1333
|
+
const registeredEvidence = ledger.claimEvidence.filter((evidence) => registeredSources.has(researchSourceVersionKey({
|
|
1334
|
+
sourceId: evidence.sourceId,
|
|
1335
|
+
uri: evidence.sourceUri,
|
|
1336
|
+
contentHash: evidence.sourceContentHash
|
|
1337
|
+
})));
|
|
1338
|
+
const materializedClaims = new Map(ledger.claims.map((claim) => [claim.id, claim]));
|
|
1339
|
+
for (const claim of ledger.claims) {
|
|
1340
|
+
assertTrackedClaimIntegrity(claim);
|
|
1341
|
+
for (const sourceUri of claim.supportingUris) if (!registeredEvidence.some((evidence) => evidence.claimId === claim.id && evidence.sourceUri === sourceUri)) throw new Error(`claim '${claim.id}' counts source '${sourceUri}' without exact registered evidence`);
|
|
1342
|
+
for (const contradictedClaimId of claim.contradicts) {
|
|
1343
|
+
const contradictedClaim = materializedClaims.get(contradictedClaimId);
|
|
1344
|
+
if (!contradictedClaim) throw new Error(`claim '${claim.id}' contradicts unmaterialized claim '${contradictedClaimId}'`);
|
|
1345
|
+
if (!contradictedClaim.contradicts.includes(claim.id)) throw new Error(`claim '${claim.id}' has an asymmetric contradiction with '${contradictedClaimId}'`);
|
|
1346
|
+
if (!registeredEvidence.some((evidence) => evidence.claimId === claim.id && evidence.contradictsClaimId === contradictedClaimId || evidence.claimId === contradictedClaimId && evidence.contradictsClaimId === claim.id)) throw new Error(`claim '${claim.id}' contradicts '${contradictedClaimId}' without exact registered evidence`);
|
|
1347
|
+
}
|
|
1348
|
+
}
|
|
1349
|
+
const claimsById = materializedClaims;
|
|
1350
|
+
const claimIds = new Set(claimsById.keys());
|
|
1351
|
+
for (const evidence of ledger.claimEvidence) {
|
|
1352
|
+
if (!registeredSources.has(researchSourceVersionKey({
|
|
1353
|
+
sourceId: evidence.sourceId,
|
|
1354
|
+
uri: evidence.sourceUri,
|
|
1355
|
+
contentHash: evidence.sourceContentHash
|
|
1356
|
+
}))) continue;
|
|
1357
|
+
const claim = claimsById.get(evidence.claimId);
|
|
1358
|
+
if (!claim?.supportingUris.includes(evidence.sourceUri)) throw new Error(`registered claim evidence '${evidence.id}' must be materialized in its claim`);
|
|
1359
|
+
if (evidence.contradictsClaimId !== void 0 && claimsById.has(evidence.contradictsClaimId) && !claim.contradicts.includes(evidence.contradictsClaimId)) throw new Error(`registered claim evidence '${evidence.id}' must materialize its contradiction`);
|
|
1360
|
+
}
|
|
1361
|
+
for (const question of ledger.questions) {
|
|
1362
|
+
assertDeepQuestionIntegrity(question);
|
|
1363
|
+
for (const claimId of question.claimIds) if (!claimIds.has(claimId)) throw new Error(`question '${question.id}' references claim '${claimId}' outside its ledger`);
|
|
1364
|
+
}
|
|
1365
|
+
}
|
|
1366
|
+
/** A ledger with nothing in it yet. */
|
|
1367
|
+
function emptyClaimLedger(id, goal) {
|
|
1368
|
+
return {
|
|
1369
|
+
schemaVersion: 2,
|
|
1370
|
+
id,
|
|
1371
|
+
...goal === void 0 ? {} : { goal },
|
|
1372
|
+
updatedAt: (/* @__PURE__ */ new Date(0)).toISOString(),
|
|
1373
|
+
rounds: 0,
|
|
1374
|
+
claimEvidence: [],
|
|
1375
|
+
registeredSources: [],
|
|
1376
|
+
claims: [],
|
|
1377
|
+
questions: []
|
|
1378
|
+
};
|
|
1379
|
+
}
|
|
1380
|
+
/**
|
|
1381
|
+
* Turn only evidence backed by a confirmed source registration into support.
|
|
1382
|
+
*
|
|
1383
|
+
* This is a monotone closure: it never removes claims or evidence, and running
|
|
1384
|
+
* it twice is a no-op. Keeping it in the ledger algebra means a source-confirming
|
|
1385
|
+
* writer and an evidence-producing writer can arrive in either order.
|
|
1386
|
+
*/
|
|
1387
|
+
function materializeRegisteredClaimEvidence(ledger) {
|
|
1388
|
+
const registered = new Set(ledger.registeredSources.map(researchSourceVersionKey));
|
|
1389
|
+
const claims = new Map(ledger.claims.map((claim) => [claim.id, claim]));
|
|
1390
|
+
for (const evidence of ledger.claimEvidence) {
|
|
1391
|
+
if (!registered.has(sourceVersionKeyOfEvidence(evidence))) continue;
|
|
1392
|
+
const host = claimSourceHost(evidence.sourceUri);
|
|
1393
|
+
const observed = {
|
|
1394
|
+
id: evidence.claimId,
|
|
1395
|
+
text: evidence.text,
|
|
1396
|
+
supportingHosts: host ? [host] : [],
|
|
1397
|
+
supportingUris: [evidence.sourceUri],
|
|
1398
|
+
contradicts: [],
|
|
1399
|
+
contested: false,
|
|
1400
|
+
firstSeenRound: evidence.firstSeenRound
|
|
1401
|
+
};
|
|
1402
|
+
const existing = claims.get(observed.id);
|
|
1403
|
+
claims.set(observed.id, existing ? mergeTrackedClaims(existing, observed) : observed);
|
|
1404
|
+
}
|
|
1405
|
+
for (const evidence of ledger.claimEvidence) {
|
|
1406
|
+
const otherId = evidence.contradictsClaimId;
|
|
1407
|
+
if (!registered.has(sourceVersionKeyOfEvidence(evidence)) || !otherId || !claims.has(otherId)) continue;
|
|
1408
|
+
const claim = claims.get(evidence.claimId);
|
|
1409
|
+
if (!claim) continue;
|
|
1410
|
+
claims.set(claim.id, {
|
|
1411
|
+
...claim,
|
|
1412
|
+
contradicts: union(claim.contradicts, [otherId]),
|
|
1413
|
+
contested: true
|
|
1414
|
+
});
|
|
1415
|
+
}
|
|
1416
|
+
return linkClaimContradictions({
|
|
1417
|
+
...ledger,
|
|
1418
|
+
claims: [...claims.values()].sort((left, right) => left.id.localeCompare(right.id))
|
|
1419
|
+
});
|
|
1420
|
+
}
|
|
1421
|
+
function sourceVersionKeyOfEvidence(evidence) {
|
|
1422
|
+
return researchSourceVersionKey({
|
|
1423
|
+
sourceId: evidence.sourceId,
|
|
1424
|
+
uri: evidence.sourceUri,
|
|
1425
|
+
contentHash: evidence.sourceContentHash
|
|
1426
|
+
});
|
|
1427
|
+
}
|
|
1428
|
+
/**
|
|
1429
|
+
* Raised when two ledgers that accumulated evidence for DIFFERENT goals are
|
|
1430
|
+
* combined. Merging them would pool two questions' evidence into one
|
|
1431
|
+
* corroboration count, which reports a claim as independently confirmed when
|
|
1432
|
+
* nobody confirmed it — strictly worse than losing the ledger, so this refuses.
|
|
1433
|
+
*/
|
|
1434
|
+
var ClaimLedgerGoalConflictError = class extends Error {
|
|
1435
|
+
ledgerId;
|
|
1436
|
+
existingGoal;
|
|
1437
|
+
incomingGoal;
|
|
1438
|
+
constructor(ledgerId, existingGoal, incomingGoal) {
|
|
1439
|
+
super(`claim ledger '${ledgerId}' accumulated evidence for goal '${existingGoal}' and cannot be merged with evidence for '${incomingGoal}'`);
|
|
1440
|
+
this.ledgerId = ledgerId;
|
|
1441
|
+
this.existingGoal = existingGoal;
|
|
1442
|
+
this.incomingGoal = incomingGoal;
|
|
1443
|
+
this.name = "ClaimLedgerGoalConflictError";
|
|
1444
|
+
}
|
|
1445
|
+
};
|
|
1446
|
+
/**
|
|
1447
|
+
* Combine two records of the same claim.
|
|
1448
|
+
*
|
|
1449
|
+
* Union on every support collection, because a claim asserted by hosts {a} in
|
|
1450
|
+
* one writer and {b} in another is asserted by two independent hosts and the
|
|
1451
|
+
* whole completion oracle turns on that count. `contested` is OR — one writer
|
|
1452
|
+
* seeing a contradiction is enough for the claim to be contested, and no later
|
|
1453
|
+
* writer that simply did not see it may clear the flag.
|
|
1454
|
+
*/
|
|
1455
|
+
function mergeTrackedClaims(base, incoming) {
|
|
1456
|
+
assertTrackedClaimIntegrity(base);
|
|
1457
|
+
assertTrackedClaimIntegrity(incoming);
|
|
1458
|
+
if (base.id !== incoming.id) throw new Error(`cannot merge claim '${base.id}' with a different claim '${incoming.id}'`);
|
|
1459
|
+
const text = incoming.firstSeenRound < base.firstSeenRound ? incoming.text : incoming.firstSeenRound > base.firstSeenRound ? base.text : incoming.text < base.text ? incoming.text : base.text;
|
|
1460
|
+
return {
|
|
1461
|
+
id: base.id,
|
|
1462
|
+
text,
|
|
1463
|
+
supportingHosts: union(base.supportingHosts, incoming.supportingHosts),
|
|
1464
|
+
supportingUris: union(base.supportingUris, incoming.supportingUris),
|
|
1465
|
+
contradicts: union(base.contradicts, incoming.contradicts),
|
|
1466
|
+
contested: base.contested || incoming.contested,
|
|
1467
|
+
firstSeenRound: Math.min(base.firstSeenRound, incoming.firstSeenRound)
|
|
1468
|
+
};
|
|
1469
|
+
}
|
|
1470
|
+
/**
|
|
1471
|
+
* Combine two ledgers for the same run.
|
|
1472
|
+
*
|
|
1473
|
+
* `addressed` on a question is OR for the same reason `contested` is: a writer
|
|
1474
|
+
* that answered a question has answered it, and a writer that never saw the
|
|
1475
|
+
* answer must not reopen it. Everything else is a union or an extreme, so this
|
|
1476
|
+
* is associative and idempotent — merge order cannot change the result and a
|
|
1477
|
+
* replayed merge is a no-op.
|
|
1478
|
+
*/
|
|
1479
|
+
function mergeClaimLedgers(base, incoming) {
|
|
1480
|
+
assertResearchClaimLedgerIntegrity(base);
|
|
1481
|
+
assertResearchClaimLedgerIntegrity(incoming);
|
|
1482
|
+
if (base.id !== incoming.id) throw new Error(`cannot merge claim ledger '${base.id}' with a different ledger '${incoming.id}'`);
|
|
1483
|
+
if (base.goal !== void 0 && incoming.goal !== void 0 && base.goal !== incoming.goal) throw new ClaimLedgerGoalConflictError(base.id, base.goal, incoming.goal);
|
|
1484
|
+
const goal = base.goal ?? incoming.goal;
|
|
1485
|
+
const claimEvidence = new Map(base.claimEvidence.map((evidence) => [evidence.id, evidence]));
|
|
1486
|
+
for (const evidence of incoming.claimEvidence) {
|
|
1487
|
+
const existing = claimEvidence.get(evidence.id);
|
|
1488
|
+
claimEvidence.set(evidence.id, existing ? mergeClaimEvidence(existing, evidence) : evidence);
|
|
1489
|
+
}
|
|
1490
|
+
const claims = new Map(base.claims.map((claim) => [claim.id, claim]));
|
|
1491
|
+
for (const claim of incoming.claims) {
|
|
1492
|
+
const existing = claims.get(claim.id);
|
|
1493
|
+
claims.set(claim.id, existing ? mergeTrackedClaims(existing, claim) : claim);
|
|
1494
|
+
}
|
|
1495
|
+
const questions = new Map(base.questions.map((question) => [question.id, question]));
|
|
1496
|
+
for (const question of incoming.questions) {
|
|
1497
|
+
const existing = questions.get(question.id);
|
|
1498
|
+
if (existing && (existing.kind !== question.kind || existing.text !== question.text)) throw new Error(`question '${question.id}' has conflicting immutable content`);
|
|
1499
|
+
questions.set(question.id, existing ? {
|
|
1500
|
+
...existing,
|
|
1501
|
+
claimIds: union(existing.claimIds, question.claimIds),
|
|
1502
|
+
addressed: existing.addressed || question.addressed,
|
|
1503
|
+
raisedRound: Math.min(existing.raisedRound, question.raisedRound)
|
|
1504
|
+
} : question);
|
|
1505
|
+
}
|
|
1506
|
+
const rounds = Math.max(base.rounds, incoming.rounds);
|
|
1507
|
+
const preparedRounds = Math.max(base.preparedRounds ?? base.rounds, incoming.preparedRounds ?? incoming.rounds);
|
|
1508
|
+
return materializeRegisteredClaimEvidence({
|
|
1509
|
+
schemaVersion: 2,
|
|
1510
|
+
id: base.id,
|
|
1511
|
+
...goal === void 0 ? {} : { goal },
|
|
1512
|
+
updatedAt: incoming.updatedAt.localeCompare(base.updatedAt) > 0 ? incoming.updatedAt : base.updatedAt,
|
|
1513
|
+
rounds,
|
|
1514
|
+
...preparedRounds > rounds ? { preparedRounds } : {},
|
|
1515
|
+
claimEvidence: [...claimEvidence.values()].sort((left, right) => left.id.localeCompare(right.id)),
|
|
1516
|
+
registeredSources: mergeSourceVersions(base.registeredSources, incoming.registeredSources),
|
|
1517
|
+
claims: [...claims.values()].sort((a, b) => a.id.localeCompare(b.id)),
|
|
1518
|
+
questions: [...questions.values()].sort((a, b) => a.id.localeCompare(b.id))
|
|
1519
|
+
});
|
|
1520
|
+
}
|
|
1521
|
+
function mergeClaimEvidence(base, incoming) {
|
|
1522
|
+
assertResearchClaimEvidenceIntegrity(base);
|
|
1523
|
+
assertResearchClaimEvidenceIntegrity(incoming);
|
|
1524
|
+
if (base.id !== incoming.id || base.claimId !== incoming.claimId || base.sourceId !== incoming.sourceId || base.sourceUri !== incoming.sourceUri || base.sourceContentHash !== incoming.sourceContentHash || base.contradictsClaimId !== incoming.contradictsClaimId) throw new Error(`claim evidence '${base.id}' has conflicting immutable content`);
|
|
1525
|
+
const text = incoming.firstSeenRound < base.firstSeenRound ? incoming.text : incoming.firstSeenRound > base.firstSeenRound ? base.text : incoming.text < base.text ? incoming.text : base.text;
|
|
1526
|
+
return {
|
|
1527
|
+
...base,
|
|
1528
|
+
text,
|
|
1529
|
+
firstSeenRound: Math.min(base.firstSeenRound, incoming.firstSeenRound)
|
|
1530
|
+
};
|
|
1531
|
+
}
|
|
1532
|
+
function mergeSourceVersions(base, incoming) {
|
|
1533
|
+
const versions = /* @__PURE__ */ new Map();
|
|
1534
|
+
for (const source of [...base, ...incoming]) {
|
|
1535
|
+
const existing = versions.get(source.sourceId);
|
|
1536
|
+
if (existing && researchSourceVersionKey(existing) !== researchSourceVersionKey(source)) throw new Error(`registered source '${source.sourceId}' has conflicting immutable content`);
|
|
1537
|
+
versions.set(source.sourceId, source);
|
|
1538
|
+
}
|
|
1539
|
+
return [...versions.values()].sort((left, right) => left.sourceId.localeCompare(right.sourceId));
|
|
1540
|
+
}
|
|
1541
|
+
/**
|
|
1542
|
+
* Make every contradiction edge symmetric and mark both ends contested.
|
|
1543
|
+
*
|
|
1544
|
+
* A contradiction is a property of a PAIR, and a writer only ever sees one side
|
|
1545
|
+
* of it: the worker that found the refuting source records "X contradicts Y" and
|
|
1546
|
+
* knows nothing about Y's record. Left one-sided, Y reads as an uncontested
|
|
1547
|
+
* claim, and the completion oracle would settle a question two sources disagree
|
|
1548
|
+
* about. `createResearchDrivingDriver` does this pairwise as it records; this is
|
|
1549
|
+
* the same rule stated over a whole ledger, for writers that assemble one from
|
|
1550
|
+
* events rather than from a live loop.
|
|
1551
|
+
*
|
|
1552
|
+
* One-sided observations stay in `claimEvidence` until both claims are backed
|
|
1553
|
+
* by registered source versions. The materialized claim projection contains
|
|
1554
|
+
* only closed pairs, so it cannot report a lone weak claim as settled merely
|
|
1555
|
+
* because its evidence named a claim that never arrived.
|
|
1556
|
+
*/
|
|
1557
|
+
function linkClaimContradictions(ledger) {
|
|
1558
|
+
const claimIds = new Set(ledger.claims.map((claim) => claim.id));
|
|
1559
|
+
const inbound = /* @__PURE__ */ new Map();
|
|
1560
|
+
for (const claim of ledger.claims) for (const other of claim.contradicts) {
|
|
1561
|
+
if (other === claim.id || !claimIds.has(other)) continue;
|
|
1562
|
+
const edges = inbound.get(other);
|
|
1563
|
+
if (edges) edges.push(claim.id);
|
|
1564
|
+
else inbound.set(other, [claim.id]);
|
|
1565
|
+
}
|
|
1566
|
+
return {
|
|
1567
|
+
...ledger,
|
|
1568
|
+
claims: ledger.claims.map((claim) => {
|
|
1569
|
+
const contradicts = union(claim.contradicts.filter((other) => other !== claim.id && claimIds.has(other)), inbound.get(claim.id) ?? []);
|
|
1570
|
+
return {
|
|
1571
|
+
...claim,
|
|
1572
|
+
contradicts,
|
|
1573
|
+
contested: contradicts.length > 0
|
|
1574
|
+
};
|
|
1575
|
+
}).sort((a, b) => a.id.localeCompare(b.id))
|
|
1576
|
+
};
|
|
1577
|
+
}
|
|
1578
|
+
/**
|
|
1579
|
+
* Set union, SORTED.
|
|
1580
|
+
*
|
|
1581
|
+
* Sorted because these collections are sets and merging must be commutative:
|
|
1582
|
+
* arrival order is not part of what the ledger says, so two writers arriving in
|
|
1583
|
+
* either order have to produce identical bytes. Preserving first-seen order
|
|
1584
|
+
* instead made `merge(a, b)` and `merge(b, a)` differ, which the order-
|
|
1585
|
+
* independence test caught — and a non-commutative merge under a filesystem
|
|
1586
|
+
* lock means the record depends on scheduling.
|
|
1587
|
+
*/
|
|
1588
|
+
function union(base, incoming) {
|
|
1589
|
+
return [.../* @__PURE__ */ new Set([...base, ...incoming])].sort();
|
|
1590
|
+
}
|
|
1591
|
+
function assertSortedUnique(label, values) {
|
|
1592
|
+
for (let index = 1; index < values.length; index += 1) if ((values[index - 1] ?? "") >= (values[index] ?? "")) throw new Error(`${label} must be sorted and contain no duplicates`);
|
|
1593
|
+
}
|
|
1594
|
+
function sameStrings(left, right) {
|
|
1595
|
+
return left.length === right.length && left.every((value, index) => value === right[index]);
|
|
1596
|
+
}
|
|
1597
|
+
//#endregion
|
|
1598
|
+
//#region src/types.ts
|
|
1599
|
+
/**
|
|
1600
|
+
* The event vocabulary, as a value so the runtime schema is DERIVED from it
|
|
1601
|
+
* rather than restated. A restated copy in `schemas.ts` drifted: it omitted
|
|
1602
|
+
* `research.iteration`, which is the only event `runVerifiedResearchLoop`
|
|
1603
|
+
* produces, so every attempt to store one would have been rejected. Nothing
|
|
1604
|
+
* caught it because nothing ever stored an event. Add a type here and the
|
|
1605
|
+
* schema accepts it in the same edit.
|
|
1606
|
+
*/
|
|
1607
|
+
const KNOWLEDGE_EVENT_TYPES = [
|
|
1608
|
+
"source.added",
|
|
1609
|
+
"proposal.applied",
|
|
1610
|
+
"index.built",
|
|
1611
|
+
"lint.run",
|
|
1612
|
+
"research.iteration",
|
|
1613
|
+
"optimization.run",
|
|
1614
|
+
"release.promoted",
|
|
1615
|
+
"release.rejected"
|
|
1616
|
+
];
|
|
1617
|
+
//#endregion
|
|
1211
1618
|
//#region src/schemas.ts
|
|
1212
1619
|
const SourceAnchorSchema = z.object({
|
|
1213
1620
|
id: z.string().min(1),
|
|
@@ -1271,20 +1678,79 @@ const KnowledgeIndexSchema = z.object({
|
|
|
1271
1678
|
});
|
|
1272
1679
|
const KnowledgeEventSchema = z.object({
|
|
1273
1680
|
id: z.string().min(1),
|
|
1274
|
-
type: z.enum(
|
|
1275
|
-
"source.added",
|
|
1276
|
-
"proposal.applied",
|
|
1277
|
-
"index.built",
|
|
1278
|
-
"lint.run",
|
|
1279
|
-
"optimization.run",
|
|
1280
|
-
"release.promoted",
|
|
1281
|
-
"release.rejected"
|
|
1282
|
-
]),
|
|
1681
|
+
type: z.enum(KNOWLEDGE_EVENT_TYPES),
|
|
1283
1682
|
createdAt: z.string().min(1),
|
|
1284
1683
|
actor: z.string().optional(),
|
|
1285
1684
|
target: z.string().optional(),
|
|
1286
1685
|
metadata: z.record(z.string(), z.unknown()).optional()
|
|
1287
1686
|
});
|
|
1687
|
+
const DeepQuestionSchema = z.object({
|
|
1688
|
+
kind: z.enum([
|
|
1689
|
+
"comparative",
|
|
1690
|
+
"mechanism",
|
|
1691
|
+
"gap",
|
|
1692
|
+
"contradiction"
|
|
1693
|
+
]),
|
|
1694
|
+
text: z.string().min(1),
|
|
1695
|
+
id: z.string().min(1),
|
|
1696
|
+
claimIds: z.array(z.string().min(1)),
|
|
1697
|
+
addressed: z.boolean(),
|
|
1698
|
+
raisedRound: z.number().int().nonnegative()
|
|
1699
|
+
}).strict().superRefine((question, context) => {
|
|
1700
|
+
reportIntegrityError(context, () => assertDeepQuestionIntegrity(question));
|
|
1701
|
+
});
|
|
1702
|
+
const ResearchClaimRecordSchema = z.object({
|
|
1703
|
+
id: z.string().min(1),
|
|
1704
|
+
text: z.string().min(1),
|
|
1705
|
+
supportingHosts: z.array(z.string().min(1)),
|
|
1706
|
+
supportingUris: z.array(z.string().min(1)),
|
|
1707
|
+
contradicts: z.array(z.string().min(1)),
|
|
1708
|
+
contested: z.boolean(),
|
|
1709
|
+
firstSeenRound: z.number().int().nonnegative()
|
|
1710
|
+
}).strict().superRefine((claim, context) => {
|
|
1711
|
+
reportIntegrityError(context, () => assertTrackedClaimIntegrity(claim));
|
|
1712
|
+
});
|
|
1713
|
+
const ResearchSourceVersionSchema = z.object({
|
|
1714
|
+
sourceId: z.string().min(1),
|
|
1715
|
+
uri: z.string().min(1),
|
|
1716
|
+
contentHash: z.string().regex(/^[a-f0-9]{64}$/)
|
|
1717
|
+
}).strict();
|
|
1718
|
+
const ResearchClaimEvidenceSchema = z.object({
|
|
1719
|
+
id: z.string().min(1),
|
|
1720
|
+
claimId: z.string().min(1),
|
|
1721
|
+
text: z.string().min(1),
|
|
1722
|
+
sourceId: z.string().min(1),
|
|
1723
|
+
sourceUri: z.string().min(1),
|
|
1724
|
+
sourceContentHash: z.string().regex(/^[a-f0-9]{64}$/),
|
|
1725
|
+
contradictsClaimId: z.string().min(1).optional(),
|
|
1726
|
+
firstSeenRound: z.number().int().nonnegative()
|
|
1727
|
+
}).strict().superRefine((evidence, context) => {
|
|
1728
|
+
reportIntegrityError(context, () => assertResearchClaimEvidenceIntegrity(evidence));
|
|
1729
|
+
});
|
|
1730
|
+
const ResearchClaimLedgerSchema = z.object({
|
|
1731
|
+
schemaVersion: z.literal(2),
|
|
1732
|
+
id: z.string().min(1),
|
|
1733
|
+
goal: z.string().trim().min(1).optional(),
|
|
1734
|
+
updatedAt: z.iso.datetime(),
|
|
1735
|
+
rounds: z.number().int().nonnegative(),
|
|
1736
|
+
preparedRounds: z.number().int().nonnegative().optional(),
|
|
1737
|
+
claimEvidence: z.array(ResearchClaimEvidenceSchema),
|
|
1738
|
+
registeredSources: z.array(ResearchSourceVersionSchema),
|
|
1739
|
+
claims: z.array(ResearchClaimRecordSchema),
|
|
1740
|
+
questions: z.array(DeepQuestionSchema)
|
|
1741
|
+
}).strict().superRefine((ledger, context) => {
|
|
1742
|
+
reportIntegrityError(context, () => assertResearchClaimLedgerIntegrity(ledger));
|
|
1743
|
+
});
|
|
1744
|
+
function reportIntegrityError(context, check) {
|
|
1745
|
+
try {
|
|
1746
|
+
check();
|
|
1747
|
+
} catch (error) {
|
|
1748
|
+
context.addIssue({
|
|
1749
|
+
code: "custom",
|
|
1750
|
+
message: error instanceof Error ? error.message : String(error)
|
|
1751
|
+
});
|
|
1752
|
+
}
|
|
1753
|
+
}
|
|
1288
1754
|
const KnowledgeBaseCandidateSchema = z.object({
|
|
1289
1755
|
id: z.string().min(1),
|
|
1290
1756
|
units: z.array(z.object({
|
|
@@ -1327,11 +1793,322 @@ const KnowledgeBaseCandidateSchema = z.object({
|
|
|
1327
1793
|
metadata: z.record(z.string(), z.unknown()).optional()
|
|
1328
1794
|
});
|
|
1329
1795
|
//#endregion
|
|
1796
|
+
//#region src/kb-store.ts
|
|
1797
|
+
/**
|
|
1798
|
+
* Where a filesystem knowledge base keeps its machine-written records, relative
|
|
1799
|
+
* to the knowledge-base root.
|
|
1800
|
+
*
|
|
1801
|
+
* This is the ONE location. `FileSystemKbStore` used to write `<dir>/index.json`
|
|
1802
|
+
* while `writeKnowledgeIndex` wrote `<root>/.agent-knowledge/index.json` — two
|
|
1803
|
+
* index writers, two files, and only the second one reachable, so a store that
|
|
1804
|
+
* had just been written to reported an empty knowledge base. Both now go through
|
|
1805
|
+
* `FileSystemKbStore`, anchored on the knowledge-base root, which is also the
|
|
1806
|
+
* directory `withKnowledgeMutation` locks: one file, one lock domain.
|
|
1807
|
+
*/
|
|
1808
|
+
const KB_STORE_DIR = ".agent-knowledge";
|
|
1809
|
+
const KB_INDEX_PATH = `${KB_STORE_DIR}/index.json`;
|
|
1810
|
+
const KB_EVENTS_PATH = `${KB_STORE_DIR}/events.json`;
|
|
1811
|
+
const KB_CLAIM_LEDGER_DIR = `${KB_STORE_DIR}/claim-ledgers`;
|
|
1812
|
+
/**
|
|
1813
|
+
* A claim-ledger id is used as a filename, so it is restricted to one safe path
|
|
1814
|
+
* segment. Rejecting rather than sanitising is deliberate: a sanitised id maps
|
|
1815
|
+
* two different runs onto one file and silently merges their belief state.
|
|
1816
|
+
*/
|
|
1817
|
+
const LEDGER_ID_PATTERN = /^[A-Za-z0-9_-][A-Za-z0-9._-]*$/;
|
|
1818
|
+
function assertClaimLedgerId(id) {
|
|
1819
|
+
if (!LEDGER_ID_PATTERN.test(id) || id === "." || id === "..") throw new Error(`claim ledger id must match ${LEDGER_ID_PATTERN} and cannot be a path segment: ${id}`);
|
|
1820
|
+
return id;
|
|
1821
|
+
}
|
|
1822
|
+
/** A URI-only ledger cannot be upgraded without re-verifying its source bytes. */
|
|
1823
|
+
var ClaimLedgerMigrationRequiredError = class extends Error {
|
|
1824
|
+
ledgerId;
|
|
1825
|
+
foundSchemaVersion;
|
|
1826
|
+
constructor(ledgerId, foundSchemaVersion) {
|
|
1827
|
+
super(`claim ledger '${ledgerId}' uses source identity schema '${foundSchemaVersion}'; archive it and re-verify its sources into a new ledger before use`);
|
|
1828
|
+
this.ledgerId = ledgerId;
|
|
1829
|
+
this.foundSchemaVersion = foundSchemaVersion;
|
|
1830
|
+
this.name = "ClaimLedgerMigrationRequiredError";
|
|
1831
|
+
}
|
|
1832
|
+
};
|
|
1833
|
+
var MemoryKbStore = class {
|
|
1834
|
+
sources = /* @__PURE__ */ new Map();
|
|
1835
|
+
pages = /* @__PURE__ */ new Map();
|
|
1836
|
+
events = [];
|
|
1837
|
+
claimLedgers = /* @__PURE__ */ new Map();
|
|
1838
|
+
index = null;
|
|
1839
|
+
async putSource(source) {
|
|
1840
|
+
this.sources.set(source.id, clone(source));
|
|
1841
|
+
}
|
|
1842
|
+
async getSource(id) {
|
|
1843
|
+
return clone(this.sources.get(id) ?? null);
|
|
1844
|
+
}
|
|
1845
|
+
async listSources() {
|
|
1846
|
+
return [...this.sources.values()].map(clone);
|
|
1847
|
+
}
|
|
1848
|
+
async putPage(page) {
|
|
1849
|
+
this.pages.set(page.id, clone(page));
|
|
1850
|
+
}
|
|
1851
|
+
async getPage(idOrPath) {
|
|
1852
|
+
return clone(this.pages.get(idOrPath) ?? [...this.pages.values()].find((page) => page.path === idOrPath) ?? null);
|
|
1853
|
+
}
|
|
1854
|
+
async listPages() {
|
|
1855
|
+
return [...this.pages.values()].map(clone);
|
|
1856
|
+
}
|
|
1857
|
+
async putIndex(index) {
|
|
1858
|
+
this.index = clone(index);
|
|
1859
|
+
}
|
|
1860
|
+
async getIndex() {
|
|
1861
|
+
if (this.index) return clone(this.index);
|
|
1862
|
+
const pages = await this.listPages();
|
|
1863
|
+
const sources = await this.listSources();
|
|
1864
|
+
return {
|
|
1865
|
+
root: "memory",
|
|
1866
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1867
|
+
sources,
|
|
1868
|
+
pages,
|
|
1869
|
+
graph: buildKnowledgeGraph(pages)
|
|
1870
|
+
};
|
|
1871
|
+
}
|
|
1872
|
+
async putEvent(event) {
|
|
1873
|
+
this.events.push(clone(event));
|
|
1874
|
+
}
|
|
1875
|
+
async listEvents(query = {}) {
|
|
1876
|
+
let out = this.events;
|
|
1877
|
+
if (query.type) out = out.filter((event) => event.type === query.type);
|
|
1878
|
+
if (query.target) out = out.filter((event) => event.target === query.target);
|
|
1879
|
+
out = [...out].sort((a, b) => a.createdAt.localeCompare(b.createdAt));
|
|
1880
|
+
return out.slice(-(query.limit ?? out.length)).map(clone);
|
|
1881
|
+
}
|
|
1882
|
+
async putClaimLedger(ledger) {
|
|
1883
|
+
const parsed = parseResearchClaimLedger(ledger);
|
|
1884
|
+
this.claimLedgers.set(assertClaimLedgerId(parsed.id), clone(parsed));
|
|
1885
|
+
}
|
|
1886
|
+
async getClaimLedger(id) {
|
|
1887
|
+
return clone(this.claimLedgers.get(assertClaimLedgerId(id)) ?? null);
|
|
1888
|
+
}
|
|
1889
|
+
async listClaimLedgers() {
|
|
1890
|
+
return [...this.claimLedgers.values()].map(clone).sort((a, b) => a.id.localeCompare(b.id));
|
|
1891
|
+
}
|
|
1892
|
+
async mergeClaimLedger(id, merge) {
|
|
1893
|
+
const key = assertClaimLedgerId(id);
|
|
1894
|
+
const next = assertMergedLedgerId(key, merge(clone(this.claimLedgers.get(key) ?? null)));
|
|
1895
|
+
await this.putClaimLedger(next);
|
|
1896
|
+
return clone(next);
|
|
1897
|
+
}
|
|
1898
|
+
};
|
|
1899
|
+
const knowledgeEventsSchema = z.array(KnowledgeEventSchema);
|
|
1900
|
+
var FileSystemKbStore = class {
|
|
1901
|
+
root;
|
|
1902
|
+
indexPath;
|
|
1903
|
+
eventsPath;
|
|
1904
|
+
claimLedgerDir;
|
|
1905
|
+
/**
|
|
1906
|
+
* A string retains the published direct-directory contract.
|
|
1907
|
+
* The object form explicitly selects a knowledge-base root and the canonical
|
|
1908
|
+
* `.agent-knowledge/` layout. A string naming that exact canonical directory
|
|
1909
|
+
* keeps its file paths but shares the root form's lock domain.
|
|
1910
|
+
*/
|
|
1911
|
+
constructor(input) {
|
|
1912
|
+
const directDirectory = typeof input === "string" ? resolve(input) : void 0;
|
|
1913
|
+
const aliasesCanonicalDirectory = directDirectory !== void 0 && basename(directDirectory) === ".agent-knowledge";
|
|
1914
|
+
this.root = aliasesCanonicalDirectory ? dirname(directDirectory) : typeof input === "string" ? input : input.root;
|
|
1915
|
+
const canonicalLayout = typeof input !== "string" || aliasesCanonicalDirectory;
|
|
1916
|
+
this.indexPath = canonicalLayout ? KB_INDEX_PATH : "index.json";
|
|
1917
|
+
this.eventsPath = canonicalLayout ? KB_EVENTS_PATH : "events.json";
|
|
1918
|
+
this.claimLedgerDir = canonicalLayout ? KB_CLAIM_LEDGER_DIR : "claim-ledgers";
|
|
1919
|
+
}
|
|
1920
|
+
async putSource(source) {
|
|
1921
|
+
const parsed = SourceRecordSchema.parse(source);
|
|
1922
|
+
await this.updateIndex((index) => ({
|
|
1923
|
+
...index,
|
|
1924
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1925
|
+
sources: [parsed, ...index.sources.filter((entry) => entry.id !== parsed.id)]
|
|
1926
|
+
}));
|
|
1927
|
+
}
|
|
1928
|
+
async getSource(id) {
|
|
1929
|
+
return withKnowledgeRead(this.root, async () => {
|
|
1930
|
+
return clone((await this.readIndex())?.sources.find((source) => source.id === id) ?? null);
|
|
1931
|
+
});
|
|
1932
|
+
}
|
|
1933
|
+
async listSources() {
|
|
1934
|
+
return withKnowledgeRead(this.root, async () => clone((await this.readIndex())?.sources ?? []));
|
|
1935
|
+
}
|
|
1936
|
+
async putPage(page) {
|
|
1937
|
+
const parsed = KnowledgePageSchema.parse(page);
|
|
1938
|
+
await this.updateIndex((index) => {
|
|
1939
|
+
const pages = [parsed, ...index.pages.filter((entry) => entry.id !== parsed.id)];
|
|
1940
|
+
return {
|
|
1941
|
+
...index,
|
|
1942
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1943
|
+
pages,
|
|
1944
|
+
graph: buildKnowledgeGraph(pages)
|
|
1945
|
+
};
|
|
1946
|
+
});
|
|
1947
|
+
}
|
|
1948
|
+
async getPage(idOrPath) {
|
|
1949
|
+
return withKnowledgeRead(this.root, async () => {
|
|
1950
|
+
return clone((await this.readIndex())?.pages.find((page) => page.id === idOrPath || page.path === idOrPath) ?? null);
|
|
1951
|
+
});
|
|
1952
|
+
}
|
|
1953
|
+
async listPages() {
|
|
1954
|
+
return withKnowledgeRead(this.root, async () => clone((await this.readIndex())?.pages ?? []));
|
|
1955
|
+
}
|
|
1956
|
+
async putIndex(index) {
|
|
1957
|
+
const parsed = KnowledgeIndexSchema.parse(index);
|
|
1958
|
+
await withKnowledgeMutation(this.root, () => writeJsonDurableWithinRoot(this.root, this.indexPath, parsed));
|
|
1959
|
+
}
|
|
1960
|
+
async getIndex() {
|
|
1961
|
+
return withKnowledgeRead(this.root, () => this.readIndex());
|
|
1962
|
+
}
|
|
1963
|
+
async putEvent(event) {
|
|
1964
|
+
const parsed = KnowledgeEventSchema.parse(event);
|
|
1965
|
+
await withKnowledgeMutation(this.root, async () => {
|
|
1966
|
+
const next = [...(await this.readEvents()).filter((entry) => entry.id !== parsed.id), parsed].sort((a, b) => a.createdAt.localeCompare(b.createdAt));
|
|
1967
|
+
await writeJsonDurableWithinRoot(this.root, this.eventsPath, next);
|
|
1968
|
+
});
|
|
1969
|
+
}
|
|
1970
|
+
async listEvents(query = {}) {
|
|
1971
|
+
return withKnowledgeRead(this.root, async () => {
|
|
1972
|
+
let events = await this.readEvents();
|
|
1973
|
+
if (query.type) events = events.filter((event) => event.type === query.type);
|
|
1974
|
+
if (query.target) events = events.filter((event) => event.target === query.target);
|
|
1975
|
+
return clone(events.slice(-(query.limit ?? events.length)));
|
|
1976
|
+
});
|
|
1977
|
+
}
|
|
1978
|
+
async putClaimLedger(ledger) {
|
|
1979
|
+
const parsed = parseResearchClaimLedger(ledger);
|
|
1980
|
+
const path = this.claimLedgerPath(parsed.id);
|
|
1981
|
+
await withKnowledgeMutation(this.root, async () => {
|
|
1982
|
+
await readJsonFile(this.root, path, researchClaimLedgerParser);
|
|
1983
|
+
await writeJsonDurableWithinRoot(this.root, path, parsed);
|
|
1984
|
+
});
|
|
1985
|
+
}
|
|
1986
|
+
async getClaimLedger(id) {
|
|
1987
|
+
const path = this.claimLedgerPath(id);
|
|
1988
|
+
return withKnowledgeRead(this.root, () => readJsonFile(this.root, path, researchClaimLedgerParser));
|
|
1989
|
+
}
|
|
1990
|
+
async listClaimLedgers() {
|
|
1991
|
+
return withKnowledgeRead(this.root, async () => {
|
|
1992
|
+
let files;
|
|
1993
|
+
try {
|
|
1994
|
+
files = await listRegularFilesWithinRoot(this.root, this.claimLedgerDir);
|
|
1995
|
+
} catch (error) {
|
|
1996
|
+
if (isMissingFile(error)) return [];
|
|
1997
|
+
throw error;
|
|
1998
|
+
}
|
|
1999
|
+
const ledgers = [];
|
|
2000
|
+
for (const file of files) {
|
|
2001
|
+
if (!file.path.endsWith(".json")) continue;
|
|
2002
|
+
ledgers.push(parseResearchClaimLedger(JSON.parse(file.bytes.toString("utf8"))));
|
|
2003
|
+
}
|
|
2004
|
+
return ledgers.sort((a, b) => a.id.localeCompare(b.id));
|
|
2005
|
+
});
|
|
2006
|
+
}
|
|
2007
|
+
async mergeClaimLedger(id, merge) {
|
|
2008
|
+
const key = assertClaimLedgerId(id);
|
|
2009
|
+
return withKnowledgeMutation(this.root, async () => {
|
|
2010
|
+
const current = await readJsonFile(this.root, this.claimLedgerPath(key), researchClaimLedgerParser);
|
|
2011
|
+
const next = assertMergedLedgerId(key, merge(current));
|
|
2012
|
+
await this.putClaimLedger(next);
|
|
2013
|
+
return next;
|
|
2014
|
+
});
|
|
2015
|
+
}
|
|
2016
|
+
async updateIndex(change) {
|
|
2017
|
+
await withKnowledgeMutation(this.root, async () => {
|
|
2018
|
+
const current = await this.readIndex() ?? emptyIndex(this.root);
|
|
2019
|
+
const next = KnowledgeIndexSchema.parse(change(current));
|
|
2020
|
+
await writeJsonDurableWithinRoot(this.root, this.indexPath, next);
|
|
2021
|
+
});
|
|
2022
|
+
}
|
|
2023
|
+
async readIndex() {
|
|
2024
|
+
return readJsonFile(this.root, this.indexPath, KnowledgeIndexSchema);
|
|
2025
|
+
}
|
|
2026
|
+
async readEvents() {
|
|
2027
|
+
return await readJsonFile(this.root, this.eventsPath, knowledgeEventsSchema) ?? [];
|
|
2028
|
+
}
|
|
2029
|
+
claimLedgerPath(id) {
|
|
2030
|
+
return `${this.claimLedgerDir}/${assertClaimLedgerId(id)}.json`;
|
|
2031
|
+
}
|
|
2032
|
+
};
|
|
2033
|
+
/**
|
|
2034
|
+
* A merge that returns a ledger under a different id would write that ledger to
|
|
2035
|
+
* the file the caller asked to merge, giving one file two identities. Refuse
|
|
2036
|
+
* rather than trust the merge function to be well behaved.
|
|
2037
|
+
*/
|
|
2038
|
+
function assertMergedLedgerId(id, merged) {
|
|
2039
|
+
if (merged.id !== id) throw new Error(`merge of claim ledger '${id}' returned a ledger with id '${merged.id}'`);
|
|
2040
|
+
return merged;
|
|
2041
|
+
}
|
|
2042
|
+
function emptyIndex(root) {
|
|
2043
|
+
return {
|
|
2044
|
+
root,
|
|
2045
|
+
generatedAt: (/* @__PURE__ */ new Date(0)).toISOString(),
|
|
2046
|
+
sources: [],
|
|
2047
|
+
pages: [],
|
|
2048
|
+
graph: {
|
|
2049
|
+
nodes: [],
|
|
2050
|
+
edges: []
|
|
2051
|
+
}
|
|
2052
|
+
};
|
|
2053
|
+
}
|
|
2054
|
+
async function readJsonFile(root, relativePath, schema) {
|
|
2055
|
+
try {
|
|
2056
|
+
const file = await readRegularFileWithinRoot(root, relativePath);
|
|
2057
|
+
return schema.parse(JSON.parse(file.bytes.toString("utf8")));
|
|
2058
|
+
} catch (error) {
|
|
2059
|
+
if (error?.code === "ENOENT") return null;
|
|
2060
|
+
throw error;
|
|
2061
|
+
}
|
|
2062
|
+
}
|
|
2063
|
+
const researchClaimLedgerParser = { parse: parseResearchClaimLedger };
|
|
2064
|
+
const legacyResearchClaimLedgerSchema = z.object({
|
|
2065
|
+
id: z.string().min(1),
|
|
2066
|
+
goal: z.string().trim().min(1).optional(),
|
|
2067
|
+
updatedAt: z.iso.datetime(),
|
|
2068
|
+
rounds: z.number().int().nonnegative(),
|
|
2069
|
+
preparedRounds: z.number().int().nonnegative().optional(),
|
|
2070
|
+
claimEvidence: z.array(z.object({
|
|
2071
|
+
id: z.string().min(1),
|
|
2072
|
+
claimId: z.string().min(1),
|
|
2073
|
+
text: z.string().min(1),
|
|
2074
|
+
sourceUri: z.string().min(1),
|
|
2075
|
+
contradictsClaimId: z.string().min(1).optional(),
|
|
2076
|
+
firstSeenRound: z.number().int().nonnegative()
|
|
2077
|
+
}).strict()),
|
|
2078
|
+
registeredSourceUris: z.array(z.string().min(1)),
|
|
2079
|
+
claims: z.array(z.object({
|
|
2080
|
+
id: z.string().min(1),
|
|
2081
|
+
text: z.string().min(1),
|
|
2082
|
+
supportingHosts: z.array(z.string().min(1)),
|
|
2083
|
+
supportingUris: z.array(z.string().min(1)),
|
|
2084
|
+
contradicts: z.array(z.string().min(1)),
|
|
2085
|
+
contested: z.boolean(),
|
|
2086
|
+
firstSeenRound: z.number().int().nonnegative()
|
|
2087
|
+
}).strict()),
|
|
2088
|
+
questions: z.array(DeepQuestionSchema)
|
|
2089
|
+
}).strict();
|
|
2090
|
+
function parseResearchClaimLedger(value) {
|
|
2091
|
+
const legacy = legacyResearchClaimLedgerSchema.safeParse(value);
|
|
2092
|
+
if (legacy.success) throw new ClaimLedgerMigrationRequiredError(legacy.data.id, "unversioned");
|
|
2093
|
+
if (isRecord(value) && value.schemaVersion !== 2 && "registeredSourceUris" in value) throw new ClaimLedgerMigrationRequiredError(typeof value.id === "string" ? value.id : "unknown", typeof value.schemaVersion === "number" ? value.schemaVersion : "unversioned");
|
|
2094
|
+
return ResearchClaimLedgerSchema.parse(value);
|
|
2095
|
+
}
|
|
2096
|
+
function isRecord(value) {
|
|
2097
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2098
|
+
}
|
|
2099
|
+
function clone(value) {
|
|
2100
|
+
return value == null ? value : JSON.parse(JSON.stringify(value));
|
|
2101
|
+
}
|
|
2102
|
+
//#endregion
|
|
1330
2103
|
//#region src/sources.ts
|
|
1331
2104
|
const sourceRegistrySchema = z.object({
|
|
1332
2105
|
generatedAt: z.string().min(1),
|
|
1333
2106
|
sources: z.array(SourceRecordSchema.passthrough())
|
|
1334
2107
|
}).strict();
|
|
2108
|
+
/** Copy and freeze an untrusted source proposal before any asynchronous work. */
|
|
2109
|
+
function snapshotSourceTextInput(input) {
|
|
2110
|
+
return deepFreeze(structuredClone(input));
|
|
2111
|
+
}
|
|
1335
2112
|
async function loadSourceRegistry(root) {
|
|
1336
2113
|
return withKnowledgeRead(root, () => loadSourceRegistryUnlocked(root));
|
|
1337
2114
|
}
|
|
@@ -1410,7 +2187,7 @@ async function preparePathSource(root, sourcePath, mode, options) {
|
|
|
1410
2187
|
};
|
|
1411
2188
|
}
|
|
1412
2189
|
async function addSourceText(root, input, options = {}) {
|
|
1413
|
-
const [record] = await commitSourceBatch(root, [await prepareTextSource(input, options)], options);
|
|
2190
|
+
const [record] = await commitSourceBatch(root, [await prepareTextSource(snapshotSourceTextInput(input), options)], options);
|
|
1414
2191
|
return record;
|
|
1415
2192
|
}
|
|
1416
2193
|
async function prepareTextSource(input, options) {
|
|
@@ -1424,9 +2201,9 @@ async function prepareTextSource(input, options) {
|
|
|
1424
2201
|
};
|
|
1425
2202
|
const adapter = (options.adapters ?? [textSourceAdapter]).find((candidate) => candidate.canLoad(adapterInput));
|
|
1426
2203
|
const loaded = adapter ? await adapter.load(adapterInput) : {};
|
|
1427
|
-
const id =
|
|
2204
|
+
const id = textSourceId(input.uri, contentHash);
|
|
1428
2205
|
const targetRel = rawSourcePath(fileName, contentHash, ".txt");
|
|
1429
|
-
const rawContent = text
|
|
2206
|
+
const rawContent = text;
|
|
1430
2207
|
return {
|
|
1431
2208
|
record: {
|
|
1432
2209
|
id,
|
|
@@ -1501,6 +2278,12 @@ function sameMutation(left, right) {
|
|
|
1501
2278
|
if (Buffer.isBuffer(left.content) && Buffer.isBuffer(right.content)) return left.content.equals(right.content);
|
|
1502
2279
|
return left.content === right.content;
|
|
1503
2280
|
}
|
|
2281
|
+
function deepFreeze(value, seen = /* @__PURE__ */ new WeakSet()) {
|
|
2282
|
+
if (typeof value !== "object" || value === null || seen.has(value)) return value;
|
|
2283
|
+
seen.add(value);
|
|
2284
|
+
for (const nested of Object.values(value)) deepFreeze(nested, seen);
|
|
2285
|
+
return Object.freeze(value);
|
|
2286
|
+
}
|
|
1504
2287
|
async function listSourceFiles(root) {
|
|
1505
2288
|
const entries = await readdir(root, { withFileTypes: true });
|
|
1506
2289
|
const out = [];
|
|
@@ -1517,7 +2300,7 @@ async function listSourceFiles(root) {
|
|
|
1517
2300
|
return out;
|
|
1518
2301
|
}
|
|
1519
2302
|
function rawSourcePath(fileName, contentHash, extension) {
|
|
1520
|
-
return join("raw", "sources", `${slugify(fileName.replace(/\.[^.]+$/, ""))}-${contentHash
|
|
2303
|
+
return join("raw", "sources", `${slugify(fileName.replace(/\.[^.]+$/, ""))}-${contentHash}${extension}`).replace(/\\/g, "/");
|
|
1521
2304
|
}
|
|
1522
2305
|
function ext(fileName) {
|
|
1523
2306
|
const idx = fileName.lastIndexOf(".");
|
|
@@ -1552,10 +2335,19 @@ async function buildKnowledgeIndexUnlocked(root) {
|
|
|
1552
2335
|
graph: buildKnowledgeGraph(pages)
|
|
1553
2336
|
};
|
|
1554
2337
|
}
|
|
2338
|
+
/**
|
|
2339
|
+
* Build the index from the knowledge tree and store it.
|
|
2340
|
+
*
|
|
2341
|
+
* The write goes through `FileSystemKbStore` rather than straight to disk: this
|
|
2342
|
+
* function and the store used to write two different index files in two
|
|
2343
|
+
* different places, so a knowledge base could hold two disagreeing indexes and
|
|
2344
|
+
* a store-based reader saw none of the indexer's work. One writer now, and it
|
|
2345
|
+
* validates through `KnowledgeIndexSchema` on the way out.
|
|
2346
|
+
*/
|
|
1555
2347
|
async function writeKnowledgeIndex(root) {
|
|
1556
2348
|
return withKnowledgeMutation(root, async () => {
|
|
1557
2349
|
const index = await buildKnowledgeIndexUnlocked(root);
|
|
1558
|
-
await
|
|
2350
|
+
await new FileSystemKbStore({ root }).putIndex(index);
|
|
1559
2351
|
return index;
|
|
1560
2352
|
});
|
|
1561
2353
|
}
|
|
@@ -1865,6 +2657,6 @@ function explainKnowledgeTarget(index, target) {
|
|
|
1865
2657
|
};
|
|
1866
2658
|
}
|
|
1867
2659
|
//#endregion
|
|
1868
|
-
export {
|
|
2660
|
+
export { mergeTrackedClaims as $, KnowledgeGraphNodeSchema as A, removeDurable as At, ClaimLedgerGoalConflictError as B, textSourceAdapter as Bt, KB_STORE_DIR as C, prepareKnowledgeFileTransaction as Ct, KnowledgeBaseCandidateSchema as D, listRegularFilesWithinRoot as Dt, DeepQuestionSchema as E, isMissingFile as Et, ResearchClaimRecordSchema as F, writeFileDurable as Ft, claimEvidenceId as G, assertResearchClaimEvidenceIntegrity as H, ResearchSourceVersionSchema as I, writeFileDurableWithinRoot as It, deepQuestionId as J, claimId as K, SourceAnchorSchema as L, writeJsonDurable as Lt, KnowledgePageSchema as M, syncDirectory as Mt, ResearchClaimEvidenceSchema as N, withSafeDescendant as Nt, KnowledgeEventSchema as O, readRegularFileNoFollow as Ot, ResearchClaimLedgerSchema as P, withSafeDirectory as Pt, mergeClaimLedgers as Q, SourceRecordSchema as R, writeJsonDurableWithinRoot as Rt, KB_INDEX_PATH as S, knowledgeFileTransactionPlanHash as St, assertClaimLedgerId as T, isKernelAnchoredPath as Tt, assertResearchClaimLedgerIntegrity as U, assertDeepQuestionIntegrity as V, assertTrackedClaimIntegrity as W, linkClaimContradictions as X, emptyClaimLedger as Y, materializeRegisteredClaimEvidence as Z, writeSourceRegistry as _, withKnowledgeMutation as _t, applyKnowledgeWriteBlocksFile as a, isScaffoldPath as at, KB_CLAIM_LEDGER_DIR as b, assertKnowledgeMutationPath as bt, validateKnowledgeIndex as c, writeJson as ct, writeKnowledgeIndex as d, normalizeLinkTarget as dt, normalizeClaimText as et, addSourcePath as f, formatFrontmatter as ft, sourceRegistryPath as g, recoverPendingKnowledgeMutation as gt, snapshotSourceTextInput as h, inspectPendingKnowledgeMutation as ht, applyKnowledgeWriteBlocks as i, initKnowledgeBase as it, KnowledgeIndexSchema as j, renameDurable as jt, KnowledgeGraphEdgeSchema as k, readRegularFileWithinRoot as kt, lintKnowledgeIndex as l, WIKILINK_REGEX as lt, loadSourceRegistry as m, acquireDurableFileLock as mt, inspectKnowledgeIndex as n, buildKnowledgeGraph as nt, isSafeKnowledgePath as o, layoutFor as ot, addSourceText as p, parseFrontmatter as pt, claimSourceHost as q, stringMetadata as r, SCAFFOLD_PAGE_BASENAMES as rt, parseKnowledgeWriteBlocks as s, loadKnowledgePages as st, explainKnowledgeTarget as t, researchSourceVersionKey as tt, buildKnowledgeIndex as u, extractWikilinks as ut, ClaimLedgerMigrationRequiredError as v, withKnowledgeRead as vt, MemoryKbStore as w, rollbackKnowledgeFileTransaction as wt, KB_EVENTS_PATH as x, finishKnowledgeFileTransaction as xt, FileSystemKbStore as y, applyKnowledgeFileTransaction as yt, KNOWLEDGE_EVENT_TYPES as z, mediaTypeFor$1 as zt };
|
|
1869
2661
|
|
|
1870
|
-
//# sourceMappingURL=inspect-
|
|
2662
|
+
//# sourceMappingURL=inspect-CJQGYuKa.js.map
|