@tangle-network/agent-knowledge 13.0.1 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,30 @@
1
1
  # Changelog
2
2
 
3
+ ## 14.0.0 — 2026-09-05
4
+
5
+ ### Changed
6
+
7
+ - `ResearchDriver` gains optional `isComplete()`.
8
+ The research loop requires this result and storage readiness before it reports ready.
9
+ An unfinished driver receives steering rounds even when storage requirements pass.
10
+ Drivers without this method retain their storage readiness behavior.
11
+ The exported interface shape requires a major release under the package compatibility check.
12
+ - Requires `agent-eval` `>=0.174.0 <0.175.0` and tests against `0.174.0`.
13
+ This cohort uses Eval's corrected complete-method result and cost accounting contracts.
14
+ - Default knowledge evaluator version `2` averages only measured dimensions.
15
+ It omits `answer_quality` without answer evaluation, `promotion_decision` without a decision, and `blocking_readiness` without blocking requirements.
16
+ Consumers must handle absent dimension keys and compare scores using their evaluator version.
17
+ Structural-only results state that no task outcome evaluation occurred.
18
+ `candidate-ready` still leaves the candidate detached from the live knowledge base.
19
+
20
+ ### Fixed
21
+
22
+ - Knowledge diagnosis runs before acquisition and updates.
23
+ One lifecycle carries findings, acquisition, and update results into final answer checks and the promotion decision.
24
+ Diagnosis alone does not consume final evaluation cases.
25
+ Disabling a required phase fails before candidate work starts.
26
+ Final measurement still uses frozen candidate bytes and cannot feed another adaptive update.
27
+
3
28
  ## 13.0.1 — 2026-09-01
4
29
 
5
30
  ### Changed
package/README.md CHANGED
@@ -309,10 +309,18 @@ Reusing a run ID with a different implementation reference fails before cached w
309
309
  Different run IDs create separate candidate workspaces, so workers can explore in parallel.
310
310
  Promotion checks the original base hash and rejects a stale candidate instead of replacing newer work.
311
311
  Candidate retries use `evaluateDevelopment` when provided, otherwise they use deterministic validation, readiness, and KB quality checks.
312
+ Diagnosis runs before acquisition and updates, using development data only.
313
+ Its findings and update results remain available to final answer checks and the promotion decision.
312
314
  Development evaluation must use only train or selection data.
313
315
  The configured `evaluate` callback and final RAG phases run once, on the first candidate that passes those development checks.
314
316
  A failed final evaluation ends the run instead of selecting another candidate against final data.
315
317
 
318
+ The default evaluator reports only measured dimensions and averages those dimensions with equal weight.
319
+ It omits `answer_quality` without answer evaluation, `promotion_decision` without a promotion decision, and `blocking_readiness` without blocking readiness requirements.
320
+ Default evaluator version `2` records this weighting.
321
+ A candidate can pass structural checks without any task outcome evaluation; the metric notes state this limit.
322
+ `candidate-ready` means the configured checks passed and the candidate remains detached from the live knowledge base.
323
+
316
324
  Candidate promotion currently requires Linux because it relies on Linux directory descriptors for exact file identity.
317
325
 
318
326
  ## Evaluate and improve RAG
@@ -1,2 +1,2 @@
1
- import { F as buildIndustryRagBenchmarkSmokeCases, H as createInMemoryBenchmarkAdapter, I as respondToIndustryMemoryBenchmarkSmokeCase, L as respondToIndustryRagBenchmarkSmokeCase, M as INDUSTRY_RAG_BENCHMARKS, N as buildFirstPartyMemoryLifecycleBenchmarkCases, P as buildIndustryMemoryBenchmarkSmokeCases, R as isKnowledgeMemoryBenchmarkCase, U as createNoopMemoryBenchmarkAdapter, a as buildKnowledgeBenchmarkScenarios, c as runKnowledgeBenchmarkSuite, d as summarizeKnowledgeBenchmarkCampaign, i as runMemoryAdapterBenchmark, j as INDUSTRY_MEMORY_BENCHMARKS, l as scoreKnowledgeBenchmarkArtifact, n as parseKnowledgeBenchmarkJsonl, o as knowledgeBenchmarkJudge, r as parseKnowledgeBenchmarkQrels, s as renderKnowledgeBenchmarkReportMarkdown, t as buildRetrievalBenchmarkCasesFromQrels, u as scoreMemoryBenchmarkArtifact } from "../benchmarks-Qk94Gnj4.js";
1
+ import { F as buildIndustryRagBenchmarkSmokeCases, H as createNoopMemoryBenchmarkAdapter, I as respondToIndustryMemoryBenchmarkSmokeCase, L as respondToIndustryRagBenchmarkSmokeCase, M as INDUSTRY_RAG_BENCHMARKS, N as buildFirstPartyMemoryLifecycleBenchmarkCases, P as buildIndustryMemoryBenchmarkSmokeCases, R as isKnowledgeMemoryBenchmarkCase, V as createInMemoryBenchmarkAdapter, a as buildKnowledgeBenchmarkScenarios, c as runKnowledgeBenchmarkSuite, d as summarizeKnowledgeBenchmarkCampaign, i as runMemoryAdapterBenchmark, j as INDUSTRY_MEMORY_BENCHMARKS, l as scoreKnowledgeBenchmarkArtifact, n as parseKnowledgeBenchmarkJsonl, o as knowledgeBenchmarkJudge, r as parseKnowledgeBenchmarkQrels, s as renderKnowledgeBenchmarkReportMarkdown, t as buildRetrievalBenchmarkCasesFromQrels, u as scoreMemoryBenchmarkArtifact } from "../benchmarks-BbPJdmHe.js";
2
2
  export { INDUSTRY_MEMORY_BENCHMARKS, INDUSTRY_RAG_BENCHMARKS, buildFirstPartyMemoryLifecycleBenchmarkCases, buildIndustryMemoryBenchmarkSmokeCases, buildIndustryRagBenchmarkSmokeCases, buildKnowledgeBenchmarkScenarios, buildRetrievalBenchmarkCasesFromQrels, createInMemoryBenchmarkAdapter, createNoopMemoryBenchmarkAdapter, isKnowledgeMemoryBenchmarkCase, knowledgeBenchmarkJudge, parseKnowledgeBenchmarkJsonl, parseKnowledgeBenchmarkQrels, renderKnowledgeBenchmarkReportMarkdown, respondToIndustryMemoryBenchmarkSmokeCase, respondToIndustryRagBenchmarkSmokeCase, runKnowledgeBenchmarkSuite, runMemoryAdapterBenchmark, scoreKnowledgeBenchmarkArtifact, scoreMemoryBenchmarkArtifact, summarizeKnowledgeBenchmarkCampaign };
@@ -4,6 +4,28 @@ import { randomUUID } from "node:crypto";
4
4
  import { join } from "node:path";
5
5
  import { canonicalJson } from "@tangle-network/agent-eval";
6
6
  import { acquireSingleRunLock, createRunCostLedger, fsCampaignStorage, resolveRunDir, runCampaign } from "@tangle-network/agent-eval/campaign";
7
+ //#region src/statistics.ts
8
+ /**
9
+ * The mean of a set of measurements.
10
+ *
11
+ * A measurement that is not a finite number is not a measurement: it is a
12
+ * scorer that divided by zero or an adapter that answered with nothing usable.
13
+ * Averaging it in would spread one broken probe across the whole aggregate,
14
+ * and carrying it through would make the aggregate itself unusable — a
15
+ * non-finite `scoreMean` fails every numeric comparison in the ranking
16
+ * comparator, so the candidate carrying it falls through to the identifier
17
+ * tie-break and places by name rather than by what it scored. Such values are
18
+ * left out, and the mean reports what was actually measured.
19
+ *
20
+ * An empty set answers `0`. That is a reported score, not an absence, and it
21
+ * is the behavior every caller in this package already depends on.
22
+ */
23
+ function mean(values) {
24
+ const finite = values.filter(Number.isFinite);
25
+ if (finite.length === 0) return 0;
26
+ return finite.reduce((sum, value) => sum + value, 0) / finite.length;
27
+ }
28
+ //#endregion
7
29
  //#region src/retrieval-eval.ts
8
30
  function retrievalConfigSurface(config) {
9
31
  return canonicalJson(config);
@@ -449,28 +471,6 @@ function rankCandidates(rows) {
449
471
  }));
450
472
  }
451
473
  //#endregion
452
- //#region src/statistics.ts
453
- /**
454
- * The mean of a set of measurements.
455
- *
456
- * A measurement that is not a finite number is not a measurement: it is a
457
- * scorer that divided by zero or an adapter that answered with nothing usable.
458
- * Averaging it in would spread one broken probe across the whole aggregate,
459
- * and carrying it through would make the aggregate itself unusable — a
460
- * non-finite `scoreMean` fails every numeric comparison in the ranking
461
- * comparator, so the candidate carrying it falls through to the identifier
462
- * tie-break and places by name rather than by what it scored. Such values are
463
- * left out, and the mean reports what was actually measured.
464
- *
465
- * An empty set answers `0`. That is a reported score, not an absence, and it
466
- * is the behavior every caller in this package already depends on.
467
- */
468
- function mean(values) {
469
- const finite = values.filter(Number.isFinite);
470
- if (finite.length === 0) return 0;
471
- return finite.reduce((sum, value) => sum + value, 0) / finite.length;
472
- }
473
- //#endregion
474
474
  //#region src/benchmarks/utils.ts
475
475
  function unique(values) {
476
476
  return [...new Set(values.filter(Boolean))];
@@ -2819,6 +2819,6 @@ function defaultDocumentTarget(documentId, targetKind) {
2819
2819
  }
2820
2820
  }
2821
2821
  //#endregion
2822
- export { reserveRecoveryAttempts as A, normalizeUsd as B, sleepForMemoryRecovery as C, hasSettledPaidCall as D, assertNoInterruptedPaidCalls as E, buildIndustryRagBenchmarkSmokeCases as F, memoryWriteResultToSourceRecord as G, createInMemoryBenchmarkAdapter as H, respondToIndustryMemoryBenchmarkSmokeCase as I, retrievalConfigFromSurface as J, buildRetrievalEvalDispatch as K, respondToIndustryRagBenchmarkSmokeCase as L, INDUSTRY_RAG_BENCHMARKS as M, buildFirstPartyMemoryLifecycleBenchmarkCases as N, readActiveAttemptJournal as O, buildIndustryMemoryBenchmarkSmokeCases as P, isKnowledgeMemoryBenchmarkCase as R, runBoundedMemoryLifecycle as S, appendDurableJournalEvent as T, createNoopMemoryBenchmarkAdapter as U, rankCandidates as V, memoryHitToSourceRecord as W, retrievalRecallJudge as X, retrievalConfigSurface as Y, scoreRetrievalArtifact as Z, MEMORY_OPERATION_CANCELLATION_TIMEOUT_MS as _, buildKnowledgeBenchmarkScenarios as a, memoryRecoveryDelayMs as b, runKnowledgeBenchmarkSuite as c, summarizeKnowledgeBenchmarkCampaign as d, acquireAgentMemoryRunLease as f, MEMORY_CAMPAIGN_DISPATCH_SHUTDOWN_TIMEOUT_MS as g, DEFAULT_MEMORY_CLEANUP_TIMEOUT_MS as h, runMemoryAdapterBenchmark as i, INDUSTRY_MEMORY_BENCHMARKS as j, reconcileInterruptedMemoryPaidCalls as k, scoreKnowledgeBenchmarkArtifact as l, AgentMemoryLifecycleUnsafeError as m, parseKnowledgeBenchmarkJsonl as n, knowledgeBenchmarkJudge as o, AgentMemoryLifecycleTimeoutError as p, partitionRetrievalScenarios as q, parseKnowledgeBenchmarkQrels as r, renderKnowledgeBenchmarkReportMarkdown as s, buildRetrievalBenchmarkCasesFromQrels as t, scoreMemoryBenchmarkArtifact as u, createBoundedMemoryAdapter as v, appendAttemptJournalEvent as w, resolveMemoryCleanupTimeoutMs as x, createMemoryExecutionPool as y, mean as z };
2822
+ export { reserveRecoveryAttempts as A, rankCandidates as B, sleepForMemoryRecovery as C, hasSettledPaidCall as D, assertNoInterruptedPaidCalls as E, buildIndustryRagBenchmarkSmokeCases as F, buildRetrievalEvalDispatch as G, createNoopMemoryBenchmarkAdapter as H, respondToIndustryMemoryBenchmarkSmokeCase as I, retrievalConfigSurface as J, partitionRetrievalScenarios as K, respondToIndustryRagBenchmarkSmokeCase as L, INDUSTRY_RAG_BENCHMARKS as M, buildFirstPartyMemoryLifecycleBenchmarkCases as N, readActiveAttemptJournal as O, buildIndustryMemoryBenchmarkSmokeCases as P, isKnowledgeMemoryBenchmarkCase as R, runBoundedMemoryLifecycle as S, appendDurableJournalEvent as T, memoryHitToSourceRecord as U, createInMemoryBenchmarkAdapter as V, memoryWriteResultToSourceRecord as W, scoreRetrievalArtifact as X, retrievalRecallJudge as Y, mean as Z, MEMORY_OPERATION_CANCELLATION_TIMEOUT_MS as _, buildKnowledgeBenchmarkScenarios as a, memoryRecoveryDelayMs as b, runKnowledgeBenchmarkSuite as c, summarizeKnowledgeBenchmarkCampaign as d, acquireAgentMemoryRunLease as f, MEMORY_CAMPAIGN_DISPATCH_SHUTDOWN_TIMEOUT_MS as g, DEFAULT_MEMORY_CLEANUP_TIMEOUT_MS as h, runMemoryAdapterBenchmark as i, INDUSTRY_MEMORY_BENCHMARKS as j, reconcileInterruptedMemoryPaidCalls as k, scoreKnowledgeBenchmarkArtifact as l, AgentMemoryLifecycleUnsafeError as m, parseKnowledgeBenchmarkJsonl as n, knowledgeBenchmarkJudge as o, AgentMemoryLifecycleTimeoutError as p, retrievalConfigFromSurface as q, parseKnowledgeBenchmarkQrels as r, renderKnowledgeBenchmarkReportMarkdown as s, buildRetrievalBenchmarkCasesFromQrels as t, scoreMemoryBenchmarkArtifact as u, createBoundedMemoryAdapter as v, appendAttemptJournalEvent as w, resolveMemoryCleanupTimeoutMs as x, createMemoryExecutionPool as y, normalizeUsd as z };
2823
2823
 
2824
- //# sourceMappingURL=benchmarks-Qk94Gnj4.js.map
2824
+ //# sourceMappingURL=benchmarks-BbPJdmHe.js.map