@remnic/core 9.3.701 → 9.3.703
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-boundary.d.ts +7 -7
- package/dist/access-boundary.js +15 -12
- package/dist/access-cli.js +36 -33
- package/dist/access-cli.js.map +1 -1
- package/dist/access-http.d.ts +7 -7
- package/dist/access-http.js +18 -15
- package/dist/access-mcp.d.ts +7 -7
- package/dist/access-mcp.js +17 -14
- package/dist/access-operations.d.ts +7 -7
- package/dist/access-operations.js +16 -13
- package/dist/{access-service-CGVWK6lZ.d.ts → access-service-COCzhgEL.d.ts} +397 -19
- package/dist/access-service.d.ts +6 -6
- package/dist/access-service.js +14 -11
- package/dist/access-surface-catalog.d.ts +7 -7
- package/dist/action-confidence.d.ts +1 -1
- package/dist/active-memory-bridge.d.ts +1 -1
- package/dist/active-memory-bridge.js +2 -2
- package/dist/active-recall.d.ts +1 -1
- package/dist/active-recall.js +7 -6
- package/dist/active-recall.js.map +1 -1
- package/dist/behavior-learner.d.ts +1 -1
- package/dist/behavior-signals.d.ts +1 -1
- package/dist/bootstrap.d.ts +5 -5
- package/dist/briefing.d.ts +1 -1
- package/dist/briefing.js +6 -5
- package/dist/buffer-surprise-report.d.ts +1 -1
- package/dist/buffer.d.ts +1 -1
- package/dist/calibration.d.ts +1 -1
- package/dist/capabilities.d.ts +1 -1
- package/dist/{catalog-DBIghceA.d.ts → catalog-DxCzhjE6.d.ts} +2 -2
- package/dist/causal-behavior.d.ts +1 -1
- package/dist/causal-consolidation.d.ts +1 -1
- package/dist/causal-consolidation.js +7 -6
- package/dist/causal-consolidation.js.map +1 -1
- package/dist/{chunk-YXLT4EMM.js → chunk-4463LUIE.js} +12 -2
- package/dist/chunk-4463LUIE.js.map +1 -0
- package/dist/{chunk-R5DB26G6.js → chunk-4SBW47WK.js} +9 -23
- package/dist/chunk-4SBW47WK.js.map +1 -0
- package/dist/{chunk-KF4TXW7Z.js → chunk-5VXVXJH6.js} +149 -6
- package/dist/{chunk-KF4TXW7Z.js.map → chunk-5VXVXJH6.js.map} +1 -1
- package/dist/{chunk-ODTWHSY2.js → chunk-64XXOQRW.js} +2 -2
- package/dist/{chunk-ZT7B64BE.js → chunk-6TAETM63.js} +2 -2
- package/dist/{chunk-HF4N43Q7.js → chunk-6VVP6NK7.js} +2 -2
- package/dist/{chunk-FN2SM5SN.js → chunk-ABCGDW3A.js} +75 -12
- package/dist/chunk-ABCGDW3A.js.map +1 -0
- package/dist/{chunk-7TAQEPLE.js → chunk-AMNLZ6SF.js} +2 -2
- package/dist/{chunk-KS7WQ4BZ.js → chunk-CZJ6QSKG.js} +3 -3
- package/dist/{chunk-27LQPUMZ.js → chunk-EOOVJK2U.js} +3 -3
- package/dist/{chunk-QP37KL5H.js → chunk-EYJD6KIO.js} +2 -2
- package/dist/{chunk-WFEZUGU5.js → chunk-EZR35XHX.js} +2 -2
- package/dist/{chunk-UU6MVCJ6.js → chunk-FIOYURII.js} +32 -25
- package/dist/chunk-FIOYURII.js.map +1 -0
- package/dist/{chunk-JO3E5VGS.js → chunk-FVI5B7DE.js} +2 -2
- package/dist/{chunk-SFMRLXIV.js → chunk-GFIARMA7.js} +15 -7
- package/dist/chunk-GFIARMA7.js.map +1 -0
- package/dist/{chunk-PONNZ54D.js → chunk-GY3SKOS4.js} +4 -4
- package/dist/{chunk-U33LWTQQ.js → chunk-HV57RHMD.js} +4 -4
- package/dist/{chunk-IKNQAGBV.js → chunk-IX3UQT4H.js} +1 -1
- package/dist/{chunk-IKNQAGBV.js.map → chunk-IX3UQT4H.js.map} +1 -1
- package/dist/{chunk-IYOPIG3E.js → chunk-IX72AAMZ.js} +2 -2
- package/dist/{chunk-JKOKX3PS.js → chunk-JMA4RYRN.js} +2 -2
- package/dist/{chunk-DEDQXIDL.js → chunk-JTKFZMZ7.js} +2 -2
- package/dist/{chunk-OLOYQZFB.js → chunk-KC6TCAWV.js} +5 -5
- package/dist/{chunk-ROZJACKP.js → chunk-KKK7YTYN.js} +4 -1
- package/dist/chunk-KKK7YTYN.js.map +1 -0
- package/dist/{chunk-EC2AYKRX.js → chunk-L6W77GWW.js} +10 -24
- package/dist/chunk-L6W77GWW.js.map +1 -0
- package/dist/{chunk-ED35D32I.js → chunk-LQ4J7ELC.js} +2 -2
- package/dist/chunk-LUPVCGYK.js +63 -0
- package/dist/chunk-LUPVCGYK.js.map +1 -0
- package/dist/{chunk-TFVVONWD.js → chunk-MBUM2Y3L.js} +2 -2
- package/dist/{chunk-T5QAZIBO.js → chunk-MOXFPLD6.js} +3 -3
- package/dist/{chunk-XTIRCSIH.js → chunk-MXEWQKM7.js} +2 -2
- package/dist/{chunk-ZYNMX6IU.js → chunk-NUIJEGVD.js} +6 -6
- package/dist/{chunk-FUCJAZ25.js → chunk-OXEAMU42.js} +529 -9
- package/dist/chunk-OXEAMU42.js.map +1 -0
- package/dist/{chunk-YO4MBK3I.js → chunk-QGJAGC2J.js} +2 -2
- package/dist/{chunk-ISLJ5WIM.js → chunk-QIMFOCSH.js} +2 -2
- package/dist/{chunk-IJEZMWKA.js → chunk-SFOAQQDJ.js} +3 -3
- package/dist/{chunk-HRUULBBV.js → chunk-SHRRWOVY.js} +81 -4
- package/dist/chunk-SHRRWOVY.js.map +1 -0
- package/dist/{chunk-NINRTFSV.js → chunk-SINGJCUR.js} +5 -5
- package/dist/{chunk-ZPQVJEVQ.js → chunk-SK2CR6MW.js} +146 -2
- package/dist/chunk-SK2CR6MW.js.map +1 -0
- package/dist/{chunk-6W2D6FGG.js → chunk-TGAHHCB6.js} +2 -2
- package/dist/{chunk-D75JXBV4.js → chunk-WN4GHSDH.js} +2 -2
- package/dist/{chunk-K4DWSPMW.js → chunk-WRGPE6AW.js} +2 -2
- package/dist/{chunk-4HIAWLA2.js → chunk-XGMCUY5P.js} +60 -114
- package/dist/chunk-XGMCUY5P.js.map +1 -0
- package/dist/{chunk-SDPDU2PM.js → chunk-YTMDF6S7.js} +2 -2
- package/dist/{chunk-MNU5G4TK.js → chunk-Z7XEIAV4.js} +2 -2
- package/dist/{chunk-HXHKLVAS.js → chunk-ZCEI242W.js} +23 -23
- package/dist/{cli-D3XeenwN.d.ts → cli-BM4xQPp4.d.ts} +3 -3
- package/dist/cli.d.ts +7 -7
- package/dist/cli.js +33 -32
- package/dist/compounding/engine.d.ts +1 -1
- package/dist/compounding/engine.js +6 -5
- package/dist/compounding/preference-consolidator.d.ts +1 -1
- package/dist/compression-optimizer.d.ts +1 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.js +4 -2
- package/dist/connectors/codex-materialize-runner.d.ts +1 -1
- package/dist/connectors/codex-materialize-runner.js +6 -5
- package/dist/connectors/codex-materialize.d.ts +1 -1
- package/dist/connectors/index.d.ts +1 -1
- package/dist/connectors/index.js +7 -6
- package/dist/consolidation-provenance-check.d.ts +1 -1
- package/dist/consolidation-undo.d.ts +1 -1
- package/dist/contradiction/index.d.ts +2 -2
- package/dist/conversation-index/backend.d.ts +1 -1
- package/dist/conversation-index/chunker.d.ts +1 -1
- package/dist/conversation-index/faiss-adapter.d.ts +1 -1
- package/dist/conversation-index/indexer.d.ts +1 -1
- package/dist/conversation-index/search.d.ts +1 -1
- package/dist/day-summary.d.ts +1 -1
- package/dist/delinearize.d.ts +1 -1
- package/dist/direct-answer-wiring.d.ts +1 -1
- package/dist/direct-answer.d.ts +1 -1
- package/dist/embedding-fallback.d.ts +1 -1
- package/dist/enrichment/index.d.ts +1 -1
- package/dist/entity-retrieval.d.ts +1 -1
- package/dist/entity-retrieval.js +6 -5
- package/dist/entity-schema.d.ts +1 -1
- package/dist/event-order-recall.js +2 -1
- package/dist/explicit-capture.d.ts +5 -5
- package/dist/explicit-capture.js +2 -2
- package/dist/explicit-cue-recall.js +2 -1
- package/dist/extraction-faithfulness.d.ts +1 -1
- package/dist/extraction-judge-telemetry.d.ts +1 -1
- package/dist/extraction-judge-training.d.ts +1 -1
- package/dist/extraction-judge.d.ts +1 -1
- package/dist/extraction.d.ts +14 -1
- package/dist/extraction.js +5 -3
- package/dist/fallback-llm.d.ts +1 -1
- package/dist/identity-continuity.d.ts +1 -1
- package/dist/importance.d.ts +1 -1
- package/dist/index.d.ts +123 -123
- package/dist/index.js +53 -52
- package/dist/index.js.map +1 -1
- package/dist/intent.d.ts +1 -1
- package/dist/lcm/engine.d.ts +1 -1
- package/dist/lcm/index.d.ts +1 -1
- package/dist/lcm/tools.d.ts +1 -1
- package/dist/lifecycle.d.ts +1 -1
- package/dist/live-connectors-runner.d.ts +1 -1
- package/dist/local-llm.d.ts +1 -1
- package/dist/maintenance/memory-governance.d.ts +1 -1
- package/dist/maintenance/memory-governance.js +6 -5
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.js +6 -5
- package/dist/maintenance/rebuild-memory-projection.js +7 -6
- package/dist/mcp-memory-inspector-app.d.ts +7 -7
- package/dist/memory-action-policy.d.ts +1 -1
- package/dist/memory-cache.d.ts +1 -1
- package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
- package/dist/memory-projection-store.d.ts +1 -1
- package/dist/memory-provenance.d.ts +1 -1
- package/dist/memory-worth-outcomes.d.ts +1 -1
- package/dist/models-json.d.ts +1 -1
- package/dist/namespaces/migrate.d.ts +2 -2
- package/dist/namespaces/migrate.js +7 -6
- package/dist/namespaces/principal.d.ts +1 -1
- package/dist/namespaces/search.d.ts +1 -1
- package/dist/namespaces/storage.d.ts +2 -2
- package/dist/namespaces/storage.js +6 -5
- package/dist/native-knowledge.d.ts +1 -1
- package/dist/operator-toolkit.d.ts +1 -1
- package/dist/operator-toolkit.js +12 -11
- package/dist/orchestration/maintenance.d.ts +2 -2
- package/dist/orchestration/maintenance.js +8 -7
- package/dist/{orchestrator-BzMCZlKn.d.ts → orchestrator-C9CDWAm6.d.ts} +4 -4
- package/dist/orchestrator.d.ts +5 -5
- package/dist/orchestrator.js +29 -27
- package/dist/patterns-cli.d.ts +1 -1
- package/dist/policy-runtime.d.ts +1 -1
- package/dist/provenance.d.ts +36 -2
- package/dist/provenance.js +5 -1
- package/dist/qmd-recall-cache.d.ts +1 -1
- package/dist/qmd.d.ts +1 -1
- package/dist/recall-disclosure-escalation.d.ts +1 -1
- package/dist/recall-explain-renderer.d.ts +1 -1
- package/dist/recall-explain-renderer.js +3 -3
- package/dist/recall-pipeline-stages.d.ts +15 -0
- package/dist/recall-pipeline-stages.js +3 -56
- package/dist/recall-pipeline-stages.js.map +1 -1
- package/dist/recall-planner-llm.d.ts +1 -1
- package/dist/recall-state.d.ts +1 -1
- package/dist/recall-tag-filter.d.ts +1 -1
- package/dist/recall-xray-cli.d.ts +1 -1
- package/dist/recall-xray-cli.js +4 -4
- package/dist/recall-xray-renderer.d.ts +1 -1
- package/dist/recall-xray-renderer.js +3 -3
- package/dist/recall-xray.d.ts +1 -1
- package/dist/recall-xray.js +2 -2
- package/dist/resolve-auth-token.d.ts +1 -1
- package/dist/response-guidance-recall.js +2 -1
- package/dist/resume-bundles.js +7 -5
- package/dist/retrieval-agents.d.ts +1 -1
- package/dist/retrieval-tiers.d.ts +1 -1
- package/dist/routing/engine.d.ts +1 -1
- package/dist/routing/store.d.ts +1 -1
- package/dist/schemas.d.ts +41 -22
- package/dist/schemas.js +1 -1
- package/dist/search/embed-helper.d.ts +1 -1
- package/dist/search/factory.d.ts +1 -1
- package/dist/search/index.d.ts +1 -1
- package/dist/search/lancedb-backend.d.ts +1 -1
- package/dist/search/meilisearch-backend.d.ts +1 -1
- package/dist/search/noop-backend.d.ts +1 -1
- package/dist/search/orama-backend.d.ts +1 -1
- package/dist/search/port.d.ts +1 -1
- package/dist/search/remote-backend.d.ts +1 -1
- package/dist/{semantic-DJR8_DMQ.d.ts → semantic-SLAa_prH.d.ts} +1 -1
- package/dist/{semantic-consolidation-BtUfv-AL.d.ts → semantic-consolidation-DgFXyALl.d.ts} +1 -1
- package/dist/semantic-consolidation.d.ts +2 -2
- package/dist/semantic-consolidation.js +7 -6
- package/dist/semantic-rule-promotion.js +6 -5
- package/dist/semantic-rule-verifier.d.ts +1 -1
- package/dist/semantic-rule-verifier.js +6 -5
- package/dist/session-observer-bands.d.ts +1 -1
- package/dist/session-observer-state.d.ts +1 -1
- package/dist/shared-context/manager.d.ts +1 -1
- package/dist/signal.d.ts +1 -1
- package/dist/storage.d.ts +13 -1
- package/dist/storage.js +5 -4
- package/dist/summarizer.d.ts +1 -1
- package/dist/summary-snapshot.d.ts +1 -1
- package/dist/targeted-fact-recall.js +2 -1
- package/dist/temporal-supersession.d.ts +1 -1
- package/dist/temporal-validity.d.ts +1 -1
- package/dist/threading.d.ts +1 -1
- package/dist/tier-migration.d.ts +1 -1
- package/dist/tier-routing.d.ts +1 -1
- package/dist/topics.d.ts +1 -1
- package/dist/transcript.d.ts +1 -1
- package/dist/transcript.js +2 -2
- package/dist/transfer/types.d.ts +12 -12
- package/dist/{types-PuiPZ9iE.d.ts → types-MWKPZnM0.d.ts} +25 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.js +1 -1
- package/dist/utility-runtime.d.ts +1 -1
- package/dist/verified-recall.js +6 -5
- package/package.json +2 -2
- package/src/access-http.ts +162 -0
- package/src/access-service.ts +215 -0
- package/src/admin/admin-surfaces.test.ts +458 -0
- package/src/admin/admin-surfaces.ts +819 -0
- package/src/event-order-recall.test.ts +53 -0
- package/src/event-order-recall.ts +61 -37
- package/src/explicit-cue-recall.ts +21 -1
- package/src/extraction.ts +91 -6
- package/src/orchestrator.ts +46 -5
- package/src/provenance-extraction.test.ts +838 -0
- package/src/provenance.ts +413 -0
- package/src/recall-pipeline-parity.test.ts +288 -0
- package/src/recall-pipeline-stages.test.ts +43 -0
- package/src/recall-pipeline-stages.ts +22 -0
- package/src/response-guidance-recall.ts +32 -36
- package/src/schemas.ts +7 -0
- package/src/storage.ts +23 -0
- package/src/targeted-fact-recall.ts +25 -32
- package/src/types.ts +24 -0
- package/dist/chunk-4HIAWLA2.js.map +0 -1
- package/dist/chunk-EC2AYKRX.js.map +0 -1
- package/dist/chunk-FN2SM5SN.js.map +0 -1
- package/dist/chunk-FUCJAZ25.js.map +0 -1
- package/dist/chunk-HRUULBBV.js.map +0 -1
- package/dist/chunk-R5DB26G6.js.map +0 -1
- package/dist/chunk-ROZJACKP.js.map +0 -1
- package/dist/chunk-SFMRLXIV.js.map +0 -1
- package/dist/chunk-UU6MVCJ6.js.map +0 -1
- package/dist/chunk-YXLT4EMM.js.map +0 -1
- package/dist/chunk-ZPQVJEVQ.js.map +0 -1
- /package/dist/{chunk-ODTWHSY2.js.map → chunk-64XXOQRW.js.map} +0 -0
- /package/dist/{chunk-ZT7B64BE.js.map → chunk-6TAETM63.js.map} +0 -0
- /package/dist/{chunk-HF4N43Q7.js.map → chunk-6VVP6NK7.js.map} +0 -0
- /package/dist/{chunk-7TAQEPLE.js.map → chunk-AMNLZ6SF.js.map} +0 -0
- /package/dist/{chunk-KS7WQ4BZ.js.map → chunk-CZJ6QSKG.js.map} +0 -0
- /package/dist/{chunk-27LQPUMZ.js.map → chunk-EOOVJK2U.js.map} +0 -0
- /package/dist/{chunk-QP37KL5H.js.map → chunk-EYJD6KIO.js.map} +0 -0
- /package/dist/{chunk-WFEZUGU5.js.map → chunk-EZR35XHX.js.map} +0 -0
- /package/dist/{chunk-JO3E5VGS.js.map → chunk-FVI5B7DE.js.map} +0 -0
- /package/dist/{chunk-PONNZ54D.js.map → chunk-GY3SKOS4.js.map} +0 -0
- /package/dist/{chunk-U33LWTQQ.js.map → chunk-HV57RHMD.js.map} +0 -0
- /package/dist/{chunk-IYOPIG3E.js.map → chunk-IX72AAMZ.js.map} +0 -0
- /package/dist/{chunk-JKOKX3PS.js.map → chunk-JMA4RYRN.js.map} +0 -0
- /package/dist/{chunk-DEDQXIDL.js.map → chunk-JTKFZMZ7.js.map} +0 -0
- /package/dist/{chunk-OLOYQZFB.js.map → chunk-KC6TCAWV.js.map} +0 -0
- /package/dist/{chunk-ED35D32I.js.map → chunk-LQ4J7ELC.js.map} +0 -0
- /package/dist/{chunk-TFVVONWD.js.map → chunk-MBUM2Y3L.js.map} +0 -0
- /package/dist/{chunk-T5QAZIBO.js.map → chunk-MOXFPLD6.js.map} +0 -0
- /package/dist/{chunk-XTIRCSIH.js.map → chunk-MXEWQKM7.js.map} +0 -0
- /package/dist/{chunk-ZYNMX6IU.js.map → chunk-NUIJEGVD.js.map} +0 -0
- /package/dist/{chunk-YO4MBK3I.js.map → chunk-QGJAGC2J.js.map} +0 -0
- /package/dist/{chunk-ISLJ5WIM.js.map → chunk-QIMFOCSH.js.map} +0 -0
- /package/dist/{chunk-IJEZMWKA.js.map → chunk-SFOAQQDJ.js.map} +0 -0
- /package/dist/{chunk-NINRTFSV.js.map → chunk-SINGJCUR.js.map} +0 -0
- /package/dist/{chunk-6W2D6FGG.js.map → chunk-TGAHHCB6.js.map} +0 -0
- /package/dist/{chunk-D75JXBV4.js.map → chunk-WN4GHSDH.js.map} +0 -0
- /package/dist/{chunk-K4DWSPMW.js.map → chunk-WRGPE6AW.js.map} +0 -0
- /package/dist/{chunk-SDPDU2PM.js.map → chunk-YTMDF6S7.js.map} +0 -0
- /package/dist/{chunk-MNU5G4TK.js.map → chunk-Z7XEIAV4.js.map} +0 -0
- /package/dist/{chunk-HXHKLVAS.js.map → chunk-ZCEI242W.js.map} +0 -0
package/src/provenance.ts
CHANGED
|
@@ -27,7 +27,9 @@ import { z } from "zod";
|
|
|
27
27
|
|
|
28
28
|
import { coerceBool, coerceNumber } from "./connectors/coerce.js";
|
|
29
29
|
import { readEnvVar } from "./runtime/env.js";
|
|
30
|
+
import { collapseWhitespace } from "./whitespace.js";
|
|
30
31
|
import type { MemoryFrontmatter, ProvenanceConfig, ProvenanceSource } from "./types.js";
|
|
32
|
+
import { isSafeMemoryContent } from "./sanitize.js";
|
|
31
33
|
|
|
32
34
|
/**
|
|
33
35
|
* Canonical key order for a serialized `ProvenanceSource` (issue #1575).
|
|
@@ -171,6 +173,71 @@ function isStrictIsoTimestamp(s: string): boolean {
|
|
|
171
173
|
);
|
|
172
174
|
}
|
|
173
175
|
|
|
176
|
+
/**
|
|
177
|
+
* Coerce a turn timestamp to a strict ISO-8601 string (cursor thread Ocveu —
|
|
178
|
+
* "Non-ISO turn timestamps drop sources"). The write-path `ProvenanceSourceSchema`
|
|
179
|
+
* rejects anything `isStrictIsoTimestamp` fails, so a turn whose `timestamp`
|
|
180
|
+
* parses via `new Date(...)` but isn't already strict ISO (e.g.
|
|
181
|
+
* `"2026/01/15 12:00:00"` or `"Jan 15 2026"`) would be silently dropped at
|
|
182
|
+
* serialization — clearing the whole `sources` array and downgrading the tag
|
|
183
|
+
* to `"none"`. Normalizing at extraction time preserves the source.
|
|
184
|
+
*
|
|
185
|
+
* Returns the strict ISO string when the timestamp already passes or
|
|
186
|
+
* `Date.parse` can round-trip it; `undefined` for empty / unparseable input
|
|
187
|
+
* so callers can decide whether to skip the source or fall back.
|
|
188
|
+
*/
|
|
189
|
+
function toStrictIsoTimestamp(ts: string | undefined | null): string | undefined {
|
|
190
|
+
if (typeof ts !== "string" || ts.length === 0) return undefined;
|
|
191
|
+
if (isStrictIsoTimestamp(ts)) return ts;
|
|
192
|
+
const parsed = Date.parse(ts);
|
|
193
|
+
if (Number.isNaN(parsed)) return undefined;
|
|
194
|
+
// Reject bare-year / numeric-only strings that Date.parse accepts (e.g.
|
|
195
|
+
// "123") — isStrictIsoTimestamp already rejected them, and the round-trip
|
|
196
|
+
// below would otherwise resurrect them. Require at least one date/time
|
|
197
|
+
// separator so a plain number never round-trips into a fake epoch.
|
|
198
|
+
if (!/[-:T]/.test(ts)) return undefined;
|
|
199
|
+
// Reject calendar-overflow dates (chatgpt-codex-connector thread dANc):
|
|
200
|
+
// Date.parse silently rolls over invalid components — "2026-02-30" becomes
|
|
201
|
+
// 2026-03-02 — fabricating the observation date. For strings that carry an
|
|
202
|
+
// explicit numeric Y/M/D prefix (the common import/provider shape), validate
|
|
203
|
+
// the calendar components are in range before accepting the shifted result.
|
|
204
|
+
// isStrictIsoTimestamp already does this for full ISO strings; this closes
|
|
205
|
+
// the gap for the non-ISO normalization path.
|
|
206
|
+
const ymd = /^(\d{4})\D(\d{1,2})\D(\d{1,2})/.exec(ts);
|
|
207
|
+
if (ymd) {
|
|
208
|
+
const [_, ys, ms, ds] = ymd;
|
|
209
|
+
const y = Number(ys), mo = Number(ms), da = Number(ds);
|
|
210
|
+
if (!isValidCalendarDate(y, mo, da)) return undefined;
|
|
211
|
+
}
|
|
212
|
+
// Also validate M/D/Y or D/M/Y formats (provider/import common shapes:
|
|
213
|
+
// 02/30/2026, 30-02-2026, etc.). Date.parse silently shifts overflow in
|
|
214
|
+
// these too. Try both month-first and day-first interpretations; if
|
|
215
|
+
// neither is a valid calendar date, reject (chatgpt-codex-connector thread
|
|
216
|
+
// dH47 — non-YMD overflow timestamps).
|
|
217
|
+
const mdy = /^(\d{1,2})\D(\d{1,2})\D(\d{4})/.exec(ts);
|
|
218
|
+
if (mdy) {
|
|
219
|
+
const a = Number(mdy[1]), b = Number(mdy[2]), yr = Number(mdy[3]);
|
|
220
|
+
if (!isValidCalendarDate(yr, a, b) && !isValidCalendarDate(yr, b, a)) {
|
|
221
|
+
return undefined;
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
const iso = new Date(parsed).toISOString();
|
|
225
|
+
return isStrictIsoTimestamp(iso) ? iso : undefined;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* Validate numeric calendar components without Date overflow (review thread
|
|
230
|
+
* dANc). Date.UTC silently normalizes Feb 30 -> Mar 2; a component round-trip
|
|
231
|
+
* catches what Date.parse accepts. Mirrors the same technique isStrictIsoTimestamp
|
|
232
|
+
* uses for full ISO strings, applied here to the non-ISO normalization path.
|
|
233
|
+
*/
|
|
234
|
+
function isValidCalendarDate(y: number, mo: number, da: number): boolean {
|
|
235
|
+
if (mo < 1 || mo > 12 || da < 1) return false;
|
|
236
|
+
const leap = (y % 4 === 0 && (y % 100 !== 0 || y % 400 === 0)) ? 29 : 28;
|
|
237
|
+
const daysInMonth = [31, leap, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
|
|
238
|
+
return da <= daysInMonth[mo - 1]!;
|
|
239
|
+
}
|
|
240
|
+
|
|
174
241
|
/**
|
|
175
242
|
* Zod schema for a single `ProvenanceSource` entry (issue #1575). Parsed
|
|
176
243
|
* JSON from frontmatter is external data, so each entry is validated here
|
|
@@ -303,3 +370,349 @@ export function parseProvenanceConfig(raw: unknown): ProvenanceConfig {
|
|
|
303
370
|
})(),
|
|
304
371
|
};
|
|
305
372
|
}
|
|
373
|
+
|
|
374
|
+
// ---------------------------------------------------------------------------
|
|
375
|
+
// Issue #1575 PR 2 — extraction-side post-parse validator.
|
|
376
|
+
// ---------------------------------------------------------------------------
|
|
377
|
+
//
|
|
378
|
+
// This runs once at write time (inside `ExtractionEngine.extract`, after the
|
|
379
|
+
// LLM output is parsed and sanitized, before the result is returned for
|
|
380
|
+
// persistence). Its job: locate each fact's LLM-provided `quote` in the
|
|
381
|
+
// buffered turn texts and build a `ProvenanceSource[]` with verified
|
|
382
|
+
// offsets. Never throws, never drops a fact (rule 34 spirit — an
|
|
383
|
+
// unverifiable span is a tagged state, not a silent failure).
|
|
384
|
+
//
|
|
385
|
+
// Reuses `collapseWhitespace` from `whitespace.ts` for normalization so there
|
|
386
|
+
// is exactly one normalizer in the codebase (issue #1575 pitfall: "do not
|
|
387
|
+
// write a second normalizer").
|
|
388
|
+
// ---------------------------------------------------------------------------
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* A turn in the buffered conversation, reduced to the fields the provenance
|
|
392
|
+
* validator needs. `ExtractionEngine.extract` maps its `BufferTurn[]` to this
|
|
393
|
+
* shape so the validator is pure and testable without the full BufferTurn type.
|
|
394
|
+
*/
|
|
395
|
+
export interface ProvenanceTurnInput {
|
|
396
|
+
content: string;
|
|
397
|
+
sessionKey?: string;
|
|
398
|
+
logicalSessionKey?: string;
|
|
399
|
+
timestamp: string;
|
|
400
|
+
turnId?: string;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/**
|
|
404
|
+
* Result of building provenance for a single extracted fact.
|
|
405
|
+
*/
|
|
406
|
+
export interface ProvenanceBuildResult {
|
|
407
|
+
/** Verified sources (one per matching turn). Absent when no quote survives. */
|
|
408
|
+
sources?: ProvenanceSource[];
|
|
409
|
+
/** Coarse strength tag persisted to frontmatter. */
|
|
410
|
+
provenance: "verified" | "unverified" | "none";
|
|
411
|
+
/**
|
|
412
|
+
* Transient signal (never persisted): `true` when `config.requireSpans`
|
|
413
|
+
* is enabled and the LLM-provided quote could not be located in ANY source
|
|
414
|
+
* turn (the strict case `ProvenanceConfig.requireSpans` documents: "facts
|
|
415
|
+
* whose quote cannot be located are routed to `pending_review`"). The
|
|
416
|
+
* extraction consumer carries this onto the in-memory `ExtractedFact` so
|
|
417
|
+
* the persist path can route the fact to the review queue. A quote that
|
|
418
|
+
* WAS located but whose source was dropped (e.g. un-coercible timestamp)
|
|
419
|
+
* does NOT set this flag — the span was found, so requireSpans is
|
|
420
|
+
* satisfied (chatgpt-codex-connector thread 4xB).
|
|
421
|
+
*/
|
|
422
|
+
requireSpansPending?: boolean;
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
/**
|
|
426
|
+
* Cap a quote string at `maxChars`, truncating at the last word boundary that
|
|
427
|
+
* fits and appending an ellipsis marker. Quotes at or under the cap pass
|
|
428
|
+
* through unchanged. Operates on Unicode code points (not UTF-16 units) so
|
|
429
|
+
* emoji and astral-plane characters are not split mid-glyph.
|
|
430
|
+
*/
|
|
431
|
+
function capQuote(quote: string, maxChars: number): string {
|
|
432
|
+
const glyphs = Array.from(quote);
|
|
433
|
+
if (glyphs.length <= maxChars) return quote;
|
|
434
|
+
// Walk backward from the cap to find a word boundary (space). If the entire
|
|
435
|
+
// span is one long token (no spaces), cut at the cap — a hard cut is better
|
|
436
|
+
// than no quote at all.
|
|
437
|
+
let cut = maxChars;
|
|
438
|
+
while (cut > 0 && !/\s/.test(glyphs[cut - 1]!)) cut--;
|
|
439
|
+
if (cut === 0) cut = maxChars; // no word boundary found
|
|
440
|
+
return glyphs.slice(0, cut).join("").trimEnd() + "\u2026";
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
/**
|
|
444
|
+
* Casefold a string for normalized matching. Uses `toLowerCase` (not
|
|
445
|
+
* `toLocaleLowerCase`) for determinism across runtime locales — the existing
|
|
446
|
+
* `normalizeFactKey` helper in extraction.ts follows the same convention.
|
|
447
|
+
*/
|
|
448
|
+
function casefold(s: string): string {
|
|
449
|
+
return s.toLowerCase();
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
/**
|
|
453
|
+
* Locate `quote` within `text`, returning whether a match was found and, when
|
|
454
|
+
* recoverable, the half-open `[start, end)` offsets. Tries exact substring
|
|
455
|
+
* first, then whitespace/case-normalized match.
|
|
456
|
+
*
|
|
457
|
+
* For the normalized path, the mapping from the normalized string back to
|
|
458
|
+
* original offsets is recovered by walking both strings in lockstep: we know
|
|
459
|
+
* `collapseWhitespace(text)` is a subsequence of `text` with whitespace runs
|
|
460
|
+
* collapsed, so we can track the original offset as we scan.
|
|
461
|
+
*
|
|
462
|
+
* Returns `{ matched: false }` when neither match succeeds. Returns
|
|
463
|
+
* `{ matched: true, offsets }` when offsets are recoverable. Returns
|
|
464
|
+
* `{ matched: true }` (offsets omitted) when the normalized substring was
|
|
465
|
+
* found but original offsets could not be recovered — callers still treat
|
|
466
|
+
* this as a verified match and record a source without offsets (cursor
|
|
467
|
+
* thread Ocver — "Normalized match drops verified provenance").
|
|
468
|
+
*/
|
|
469
|
+
type LocateQuoteResult =
|
|
470
|
+
| { matched: false }
|
|
471
|
+
| { matched: true; offsets?: { charStart: number; charEnd: number } };
|
|
472
|
+
|
|
473
|
+
function locateQuoteOffsets(quote: string, text: string): LocateQuoteResult {
|
|
474
|
+
// 1. Exact substring match (handles unicode, curly quotes, emoji verbatim).
|
|
475
|
+
const exactIdx = text.indexOf(quote);
|
|
476
|
+
if (exactIdx >= 0) {
|
|
477
|
+
return { matched: true, offsets: { charStart: exactIdx, charEnd: exactIdx + quote.length } };
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
// 2. Whitespace/case-normalized match. Collapse runs of whitespace and
|
|
481
|
+
// casefold both sides, then find the normalized quote in the normalized
|
|
482
|
+
// text. Recover original offsets by scanning forward from the normalized
|
|
483
|
+
// match start through the original text, accumulating non-whitespace
|
|
484
|
+
// glyphs until we've consumed the normalized quote length.
|
|
485
|
+
const normQuote = collapseWhitespace(casefold(quote));
|
|
486
|
+
if (normQuote.length === 0) return { matched: false };
|
|
487
|
+
const normText = collapseWhitespace(casefold(text));
|
|
488
|
+
const normIdx = normText.indexOf(normQuote);
|
|
489
|
+
if (normIdx < 0) return { matched: false };
|
|
490
|
+
|
|
491
|
+
// Recover original offsets: walk the original text, skipping leading
|
|
492
|
+
// whitespace to align with the collapsed form, then track how many
|
|
493
|
+
// normalized chars we've consumed.
|
|
494
|
+
const normQuoteLen = normQuote.length;
|
|
495
|
+
let origIdx = 0;
|
|
496
|
+
// Skip leading whitespace in original text (collapseWhitespace trims both ends).
|
|
497
|
+
while (origIdx < text.length && /\s/.test(text[origIdx]!)) origIdx++;
|
|
498
|
+
// Walk through original text, counting normalized chars consumed.
|
|
499
|
+
// Each non-whitespace char in original = 1 normalized char.
|
|
500
|
+
// Whitespace runs in original = 1 space in normalized (but only between non-ws).
|
|
501
|
+
let normPos = 0;
|
|
502
|
+
let origStart = -1;
|
|
503
|
+
let origEnd = -1;
|
|
504
|
+
let prevWasWs = true; // suppress the leading space in collapsed form
|
|
505
|
+
while (origIdx < text.length && origEnd < 0) {
|
|
506
|
+
const ch = text[origIdx]!;
|
|
507
|
+
if (/\s/.test(ch)) {
|
|
508
|
+
if (!prevWasWs) {
|
|
509
|
+
// This whitespace run collapses to a single space in normalized text.
|
|
510
|
+
if (normPos === normIdx) origStart = origIdx; // normalized space aligns
|
|
511
|
+
normPos++;
|
|
512
|
+
prevWasWs = true;
|
|
513
|
+
}
|
|
514
|
+
origIdx++;
|
|
515
|
+
continue;
|
|
516
|
+
}
|
|
517
|
+
// Non-whitespace char.
|
|
518
|
+
if (normPos === normIdx && origStart < 0) {
|
|
519
|
+
origStart = origIdx;
|
|
520
|
+
}
|
|
521
|
+
normPos++;
|
|
522
|
+
prevWasWs = false;
|
|
523
|
+
origIdx++;
|
|
524
|
+
if (normPos >= normIdx + normQuoteLen) {
|
|
525
|
+
origEnd = origIdx;
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
if (origStart >= 0 && origEnd >= 0) {
|
|
529
|
+
return { matched: true, offsets: { charStart: origStart, charEnd: origEnd } };
|
|
530
|
+
}
|
|
531
|
+
// Normalized substring was found, but original offsets are not recoverable
|
|
532
|
+
// (edge case in the normalized mapping). The match still counts as verified
|
|
533
|
+
// — record a source without offsets rather than dropping the turn entirely
|
|
534
|
+
// (cursor thread Ocver). charStart/charEnd are best-effort debugging aids,
|
|
535
|
+
// not a precondition for a verified source.
|
|
536
|
+
return { matched: true };
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Build provenance sources for a single extracted fact by locating its
|
|
541
|
+
* LLM-provided `quote` in the buffered turns (issue #1575 PR 2).
|
|
542
|
+
*
|
|
543
|
+
* Matching strategy (per issue design):
|
|
544
|
+
* 1. Exact substring match in a turn → `provenance: "verified"` with
|
|
545
|
+
* `charStart`/`charEnd` offsets.
|
|
546
|
+
* 2. Whitespace/case-normalized match → `"verified"`, offsets recovered
|
|
547
|
+
* when possible, else omitted.
|
|
548
|
+
* 3. No match in any turn → `"unverified"` — the quote survives as a
|
|
549
|
+
* source (the LLM vouched for it) but without located offsets.
|
|
550
|
+
* 4. No quote provided by the LLM → `"none"` — no evidence to record.
|
|
551
|
+
*
|
|
552
|
+
* A quote appearing in multiple turns produces multiple sources (one per
|
|
553
|
+
* matching turn) so a repeated utterance backs the fact from each occurrence.
|
|
554
|
+
*
|
|
555
|
+
* The quote is capped at `config.maxQuoteChars` (truncated at a word boundary
|
|
556
|
+
* with an ellipsis marker) BEFORE storage. Locating uses the original
|
|
557
|
+
* (untruncated) quote for maximum match fidelity; the capped excerpt is what
|
|
558
|
+
* gets persisted.
|
|
559
|
+
*
|
|
560
|
+
* When `config.enabled === false`, returns `{ provenance: "none" }`
|
|
561
|
+
* immediately — byte-identical to pre-feature extraction (rule 39).
|
|
562
|
+
*
|
|
563
|
+
* Never throws. An unexpected error degrades to `{ provenance: "none" }` so
|
|
564
|
+
* extraction never crashes on a provenance hiccup (rule 13/18).
|
|
565
|
+
*/
|
|
566
|
+
/**
|
|
567
|
+
* Strip a leading extraction-prompt role label from a quote (cursor thread Oc3Z2
|
|
568
|
+
* — "Quote prompt mismatches validator"). The extraction prompt renders each
|
|
569
|
+
* buffered turn as `[role] content` (or `[context role] content`), so a
|
|
570
|
+
* faithful LLM may include that prefix in its verbatim quote. buildFactProvenance
|
|
571
|
+
* searches the raw `turn.content` (no prefix), so a quote carrying the label
|
|
572
|
+
* would never match. Stripping the leading label lets the actual utterance
|
|
573
|
+
* verify against the turn text. Only a SINGLE leading label is stripped — a
|
|
574
|
+
* multi-turn quote (with embedded labels) is left untouched and will simply
|
|
575
|
+
* fail to match a single turn, as before.
|
|
576
|
+
*
|
|
577
|
+
* The regex is constrained to the labels the prompt actually emits — `user`,
|
|
578
|
+
* `assistant`, optionally prefixed with `context ` (extraction.ts renders
|
|
579
|
+
* `[user]`, `[assistant]`, `[context user]`, `[context assistant]`). A
|
|
580
|
+
* real utterance that happens to start with other bracketed text (e.g.
|
|
581
|
+
* `[do not] deploy before approval`, `[P1] fix the cache`) is preserved
|
|
582
|
+
* verbatim so the quote can match the turn and the persisted span retains its
|
|
583
|
+
* original meaning (chatgpt-codex-connector thread 4xA — limiting role-prefix
|
|
584
|
+
* stripping to actual prompt labels).
|
|
585
|
+
*/
|
|
586
|
+
function stripLeadingRolePrefix(quote: string): string {
|
|
587
|
+
return quote.replace(/^\s*\[(?:context\s+)?(?:user|assistant)\]\s+/i, "");
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
export function buildFactProvenance(
|
|
591
|
+
factQuote: string | null | undefined,
|
|
592
|
+
turns: ReadonlyArray<ProvenanceTurnInput>,
|
|
593
|
+
config: ProvenanceConfig,
|
|
594
|
+
): ProvenanceBuildResult {
|
|
595
|
+
if (!config.enabled) return { provenance: "none" };
|
|
596
|
+
const rawQuote = typeof factQuote === "string" ? factQuote.trim() : "";
|
|
597
|
+
// requireSpans (chatgpt-codex-connector thread dEsu + cursor thread dGKJ):
|
|
598
|
+
// every early exit that drops a fact's span (no quote, empty after label
|
|
599
|
+
// strip, or unsafe quote) must flag requireSpansPending when the operator
|
|
600
|
+
// opted into requireSpans, so the persist path routes the fact to
|
|
601
|
+
// pending_review instead of active. Only the disabled-feature exit (above)
|
|
602
|
+
// and the unexpected-error catch (below) omit the flag — those are
|
|
603
|
+
// policy-neutral degradations, not a missing-span decision.
|
|
604
|
+
const noneResult = (): ProvenanceBuildResult =>
|
|
605
|
+
config.requireSpans === true
|
|
606
|
+
? { provenance: "none", requireSpansPending: true }
|
|
607
|
+
: { provenance: "none" };
|
|
608
|
+
if (rawQuote.length === 0) return noneResult();
|
|
609
|
+
// Strip a leading prompt role label so a faithful quote verifies against
|
|
610
|
+
// the raw turn content (cursor thread Oc3Z2). Prefer the RAW quote when it
|
|
611
|
+
// matches at least one turn — an utterance that literally begins with
|
|
612
|
+
// [user]/[assistant] must verify as-is, not be truncated to the post-label
|
|
613
|
+
// text (cursor/codex thread dEsw). The strip handles the common case where
|
|
614
|
+
// the LLM includes the prompt label; this preserves the rare case where the
|
|
615
|
+
// utterance itself starts with that text.
|
|
616
|
+
const strippedQuote = stripLeadingRolePrefix(rawQuote);
|
|
617
|
+
const useRawQuote =
|
|
618
|
+
rawQuote === strippedQuote ||
|
|
619
|
+
turns.some(
|
|
620
|
+
(t) =>
|
|
621
|
+
typeof t?.content === "string" &&
|
|
622
|
+
t.content.length > 0 &&
|
|
623
|
+
locateQuoteOffsets(rawQuote, t.content).matched,
|
|
624
|
+
);
|
|
625
|
+
const quote = useRawQuote ? rawQuote : strippedQuote;
|
|
626
|
+
if (quote.length === 0) return noneResult();
|
|
627
|
+
// Sanitize the quote before persisting it as a provenance span
|
|
628
|
+
// (chatgpt-codex-connector thread dANZ): the fact body is sanitized via
|
|
629
|
+
// sanitizeMemoryContent, but sources[].quote was persisted verbatim,
|
|
630
|
+
// reintroducing unsafe memory text (e.g. "ignore previous instructions")
|
|
631
|
+
// through the provenance field exposed to memory_get/x-ray/faithfulness.
|
|
632
|
+
// An unsafe quote cannot serve as evidence — drop the source entirely
|
|
633
|
+
// (consistent with how the body is sanitized: unsafe text is redacted,
|
|
634
|
+
// and a redacted quote is useless as a verbatim span).
|
|
635
|
+
if (!isSafeMemoryContent(quote)) return noneResult();
|
|
636
|
+
|
|
637
|
+
try {
|
|
638
|
+
// Search every turn for the quote. Collect verified sources.
|
|
639
|
+
const sources: ProvenanceSource[] = [];
|
|
640
|
+
// Track the first turn where the quote was located even when its
|
|
641
|
+
// timestamp can't be coerced (cursor thread 4Pj — "Bad timestamp drops
|
|
642
|
+
// matched source"). Without this, a located-but-unverifiable quote falls
|
|
643
|
+
// through to the unverified branch and is attributed to the *last* turn's
|
|
644
|
+
// session, mislabeling the source's origin session. Also drives the
|
|
645
|
+
// requireSpans signal: only a quote that was NOT located in any turn
|
|
646
|
+
// qualifies for pending_review routing under requireSpans (thread 4xB).
|
|
647
|
+
let locatedTurn: ProvenanceTurnInput | undefined;
|
|
648
|
+
for (const turn of turns) {
|
|
649
|
+
if (!turn || typeof turn.content !== "string" || turn.content.length === 0) continue;
|
|
650
|
+
const located = locateQuoteOffsets(quote, turn.content);
|
|
651
|
+
// cursor thread Ocver: a normalized match counts as verified even when
|
|
652
|
+
// original offsets are unrecoverable — record the source without
|
|
653
|
+
// charStart/charEnd instead of skipping the turn.
|
|
654
|
+
if (!located.matched) continue;
|
|
655
|
+
if (!locatedTurn) locatedTurn = turn;
|
|
656
|
+
// cursor thread Ocveu: normalize the turn timestamp to strict ISO so
|
|
657
|
+
// the write-path ProvenanceSourceSchema keeps the source. Skip the turn
|
|
658
|
+
// when the timestamp can't be coerced — pushing it would guarantee a
|
|
659
|
+
// serialization drop (and the tag downgrade) for this source.
|
|
660
|
+
const observedAt = toStrictIsoTimestamp(turn.timestamp);
|
|
661
|
+
if (!observedAt) continue;
|
|
662
|
+
sources.push({
|
|
663
|
+
sessionKey: turn.sessionKey ?? turn.logicalSessionKey ?? "unknown",
|
|
664
|
+
...(turn.turnId ? { turnId: turn.turnId } : {}),
|
|
665
|
+
observedAt,
|
|
666
|
+
quote: capQuote(quote, config.maxQuoteChars),
|
|
667
|
+
...(located.offsets
|
|
668
|
+
? { charStart: located.offsets.charStart, charEnd: located.offsets.charEnd }
|
|
669
|
+
: {}),
|
|
670
|
+
});
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
if (sources.length > 0) {
|
|
674
|
+
return { sources, provenance: "verified" };
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
// Quote provided but could not be turned into a verified source.
|
|
678
|
+
// Determine the session to attribute: prefer the turn where the quote was
|
|
679
|
+
// LOCATED even if its timestamp couldn't be coerced (cursor thread 4Pj — a
|
|
680
|
+
// located quote must not be tied to the *last* turn's session). If the
|
|
681
|
+
// quote was not located in any turn at all, fall back to the last turn's
|
|
682
|
+
// session (the documented unverified behavior: the LLM vouched for the
|
|
683
|
+
// excerpt but we couldn't pin it to a character offset). Normalize the
|
|
684
|
+
// timestamp (thread Ocveu); fall back to epoch when no turn supplies a
|
|
685
|
+
// coercible timestamp so the unverified source still survives the
|
|
686
|
+
// write-path schema.
|
|
687
|
+
const fallbackTurn = locatedTurn ?? turns[turns.length - 1];
|
|
688
|
+
const fallbackSessionKey = fallbackTurn
|
|
689
|
+
? (fallbackTurn.sessionKey ?? fallbackTurn.logicalSessionKey ?? "unknown")
|
|
690
|
+
: "unknown";
|
|
691
|
+
const fallbackObservedAt = fallbackTurn
|
|
692
|
+
? (toStrictIsoTimestamp(fallbackTurn.timestamp) ?? new Date(0).toISOString())
|
|
693
|
+
: new Date(0).toISOString();
|
|
694
|
+
// requireSpans signal (chatgpt-codex-connector thread 4xB): when an
|
|
695
|
+
// operator opts into provenance.requireSpans, a fact whose quote could
|
|
696
|
+
// not be located in ANY turn (locatedTurn undefined) is flagged so the
|
|
697
|
+
// persist path routes it to pending_review instead of active. A quote
|
|
698
|
+
// that WAS located (locatedTurn set) satisfies requireSpans even when
|
|
699
|
+
// its source was dropped for an un-coercible timestamp — the span was
|
|
700
|
+
// found, so the fact has the grounding requireSpans demands.
|
|
701
|
+
const requireSpansPending =
|
|
702
|
+
config.requireSpans === true && locatedTurn === undefined;
|
|
703
|
+
return {
|
|
704
|
+
sources: [
|
|
705
|
+
{
|
|
706
|
+
sessionKey: fallbackSessionKey,
|
|
707
|
+
observedAt: fallbackObservedAt,
|
|
708
|
+
quote: capQuote(quote, config.maxQuoteChars),
|
|
709
|
+
},
|
|
710
|
+
],
|
|
711
|
+
provenance: "unverified",
|
|
712
|
+
...(requireSpansPending ? { requireSpansPending: true } : {}),
|
|
713
|
+
};
|
|
714
|
+
} catch {
|
|
715
|
+
// Never crash extraction on a provenance error (rule 13/18).
|
|
716
|
+
return { provenance: "none" };
|
|
717
|
+
}
|
|
718
|
+
}
|