@origintrail-official/dkg-core 10.0.0 → 10.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/dist/crypto/canonicalize.d.ts.map +1 -1
  2. package/dist/crypto/canonicalize.js +6 -1
  3. package/dist/crypto/canonicalize.js.map +1 -1
  4. package/dist/crypto/index.d.ts +1 -0
  5. package/dist/crypto/index.d.ts.map +1 -1
  6. package/dist/crypto/index.js +1 -0
  7. package/dist/crypto/index.js.map +1 -1
  8. package/dist/crypto/term-canon.d.ts +2 -0
  9. package/dist/crypto/term-canon.d.ts.map +1 -0
  10. package/dist/crypto/term-canon.js +667 -0
  11. package/dist/crypto/term-canon.js.map +1 -0
  12. package/dist/ensure-dkg-node-config.d.ts +27 -0
  13. package/dist/ensure-dkg-node-config.d.ts.map +1 -1
  14. package/dist/ensure-dkg-node-config.js +55 -2
  15. package/dist/ensure-dkg-node-config.js.map +1 -1
  16. package/dist/errors.d.ts +18 -0
  17. package/dist/errors.d.ts.map +1 -1
  18. package/dist/errors.js +21 -0
  19. package/dist/errors.js.map +1 -1
  20. package/dist/index.d.ts +7 -3
  21. package/dist/index.d.ts.map +1 -1
  22. package/dist/index.js +6 -2
  23. package/dist/index.js.map +1 -1
  24. package/dist/log-redaction.d.ts +45 -0
  25. package/dist/log-redaction.d.ts.map +1 -0
  26. package/dist/log-redaction.js +163 -0
  27. package/dist/log-redaction.js.map +1 -0
  28. package/dist/logger.d.ts +18 -2
  29. package/dist/logger.d.ts.map +1 -1
  30. package/dist/logger.js +23 -4
  31. package/dist/logger.js.map +1 -1
  32. package/dist/proto/index.d.ts +1 -1
  33. package/dist/proto/index.d.ts.map +1 -1
  34. package/dist/proto/index.js +1 -1
  35. package/dist/proto/index.js.map +1 -1
  36. package/dist/proto/storage-ack.d.ts +9 -0
  37. package/dist/proto/storage-ack.d.ts.map +1 -1
  38. package/dist/proto/storage-ack.js +13 -0
  39. package/dist/proto/storage-ack.js.map +1 -1
  40. package/dist/protocol-router.d.ts +1 -0
  41. package/dist/protocol-router.d.ts.map +1 -1
  42. package/dist/protocol-router.js +20 -0
  43. package/dist/protocol-router.js.map +1 -1
  44. package/dist/publisher-extension.d.ts +0 -12
  45. package/dist/publisher-extension.d.ts.map +1 -1
  46. package/dist/publisher-extension.js +7 -7
  47. package/dist/publisher-extension.js.map +1 -1
  48. package/dist/rdf-literal-size.d.ts +49 -0
  49. package/dist/rdf-literal-size.d.ts.map +1 -0
  50. package/dist/rdf-literal-size.js +111 -0
  51. package/dist/rdf-literal-size.js.map +1 -0
  52. package/dist/setup-network.d.ts +64 -0
  53. package/dist/setup-network.d.ts.map +1 -0
  54. package/dist/setup-network.js +58 -0
  55. package/dist/setup-network.js.map +1 -0
  56. package/dist/telemetry-api.d.ts +89 -0
  57. package/dist/telemetry-api.d.ts.map +1 -0
  58. package/dist/telemetry-api.js +124 -0
  59. package/dist/telemetry-api.js.map +1 -0
  60. package/package.json +4 -1
@@ -0,0 +1,667 @@
1
+ // Protocol-defined, BACKEND-INDEPENDENT canonicalization of an RDF object term,
2
+ // applied at the V10 merkle leaf (tripleContentV10) so EVERY node computes the
3
+ // identical leaf for the same triple regardless of which triple store it runs
4
+ // (oxigraph, blazegraph, a SPARQL endpoint, future backends) and which version.
5
+ // Without this the leaf delegated literal canonicalization to whatever string the
6
+ // backend emitted, so a publisher sealing pre-store and a peer recomputing
7
+ // post-store could hash different serializations of the SAME triple →
8
+ // MERKLE_MISMATCH_IN_SWM (okf-dkg-vm-validation-report.md), and two nodes on
9
+ // different backends could fork RandomSampling (the contract hashes this exact
10
+ // content: leaf = keccak256(content)).
11
+ //
12
+ // The canonical form is DEFINED to equal the value-space canonicalization the
13
+ // network already deploys (oxigraph 0.5.5), verified byte-for-byte by the
14
+ // oxigraph-oracle test (packages/publisher/test/term-canon-oracle.test.ts), so it
15
+ // is the IDENTITY on already-canonical (store-loaded) terms ⇒ no migration; a
16
+ // coordinated release suffices.
17
+ //
18
+ // STRUCTURE (per consensus primitive: easy to audit, hard to drift):
19
+ // parse → validate against oxigraph's EXACT accepted set → canonicalize if
20
+ // valid, else return the escaping-normalized term VERBATIM. Every normalize is
21
+ // gated by a validate: oxigraph keeps an ill-typed-but-syntactically-odd literal
22
+ // verbatim (e.g. "not-a-date+00:00"^^xsd:dateTime), so we must NOT mutate it.
23
+ //
24
+ // Covered (all verified against oxigraph 0.5.5 — see the oracle test):
25
+ // - literal-content ESCAPING (decode N-Triples escapes, re-emit oxigraph's
26
+ // minimal \ " \n \r escaping); BARE datatype IRIs ("v"^^IRI w/o <>) folded to
27
+ // the bracketed form; language-tag lowercasing; xsd:string elision.
28
+ // - xsd:integer family (collapse→xsd:integer iff value∈i64; xsd:integer
29
+ // arbitrary precision); xsd:decimal; xsd:boolean; xsd:double / xsd:float.
30
+ // - the date/time family with FULL value-space validation + the +00:00/-00:00→Z
31
+ // tz fold (xsd:dateTime/time/date/gYear/gYearMonth/gMonthDay/gMonth/gDay) and
32
+ // the T24:00→next-day roll (oxigraph rolls iff minute==0 OR seconds==0).
33
+ // - xsd:duration / dayTimeDuration / yearMonthDuration: FULL value-space
34
+ // normalization — leading-zero strip + component carry/overflow (months as
35
+ // i64, seconds as a 10^18-scaled i128 fixed-point, mirroring oxsdatatypes),
36
+ // subtype-component constraints, all-zero → PT0S / P0M.
37
+ // Other datatypes (hexBinary, base64Binary, anyURI, token, custom IRIs, …) are
38
+ // returned with normalized escaping but otherwise verbatim — matching oxigraph.
39
+ const XSD = 'http://www.w3.org/2001/XMLSchema#';
40
+ const XSD_STRING = XSD + 'string';
41
+ const XSD_INTEGER = XSD + 'integer';
42
+ const INTEGER_TYPES = new Set([
43
+ 'integer', 'int', 'long', 'short', 'byte',
44
+ 'nonNegativeInteger', 'positiveInteger', 'nonPositiveInteger', 'negativeInteger',
45
+ 'unsignedLong', 'unsignedInt', 'unsignedShort', 'unsignedByte',
46
+ ].map((t) => XSD + t));
47
+ const DURATION_TYPES = new Set(['duration', 'dayTimeDuration', 'yearMonthDuration'].map((t) => XSD + t));
48
+ // oxigraph parses integers (incl. xsd:integer) into a signed 64-bit int and only
49
+ // THEN re-serializes canonically; a value outside i64 fails to parse and is kept
50
+ // VERBATIM (sign + leading zeros preserved) — for EVERY integer type, xsd:integer
51
+ // included. So canonicalization (collapse-to-integer + strip sign/zeros) applies
52
+ // iff the value fits i64.
53
+ const I64_MIN = -9223372036854775808n;
54
+ const I64_MAX = 9223372036854775807n;
55
+ // oxsdatatypes Duration: months is i64, seconds is a Decimal == i128 scaled by
56
+ // 10^18. A duration whose normalized months/seconds overflow these is rejected by
57
+ // oxigraph (kept verbatim), so we must reject (→ verbatim) at the same boundary.
58
+ const I128_MIN = -(1n << 127n);
59
+ const I128_MAX = (1n << 127n) - 1n;
60
+ const DEC_SCALE = 10n ** 18n;
61
+ // A literal is "<lex>"(@tag | ^^<IRI> | ^^IRI). N-Triples mandates the bracketed
62
+ // datatype, but oxigraph's storage adapter also accepts the BARE "v"^^IRI form on
63
+ // input and canonicalizes it to "v"^^<IRI>, so we accept (m[3] bracketed, m[4]
64
+ // bare) and always re-emit bracketed. The bare alternative is `[^<]…` so a
65
+ // MALFORMED bracketed datatype (e.g. "a"^^<b>extra, "x"^^<>) does NOT get
66
+ // mis-captured as a bare IRI — it fails the match and is returned verbatim, as
67
+ // oxigraph does.
68
+ const RE_LITERAL = /^"((?:[^"\\]|\\.)*)"(?:@([A-Za-z0-9-]+)|\^\^(?:<([^>]+)>|([^<].*)))?$/;
69
+ export function canonicalizeObjectTermForHash(object) {
70
+ if (object.length === 0 || object.charCodeAt(0) !== 34 /* " */)
71
+ return object; // IRI / blank / genid
72
+ const m = RE_LITERAL.exec(object);
73
+ if (!m)
74
+ return object;
75
+ const lang = m[2];
76
+ // oxigraph decodes N-Triples UCHAR (\uXXXX / \UXXXXXXXX) escapes inside the
77
+ // datatype IRI on parse, so the canonical form (and any datatype matching below)
78
+ // must run on the decoded IRI — e.g. <…XMLSchema#integer> ≡ xsd:integer.
79
+ const dtRaw = m[3] ?? m[4];
80
+ const dt = dtRaw === undefined ? undefined : decodeIriEscapes(dtRaw);
81
+ // Literal CONTENT escaping is normalized for every literal (a store decodes
82
+ // \uXXXX / \t / \U… to raw UTF-8 and re-emits only \ " \n \r escaped).
83
+ const lex = normalizeEscaping(m[1]);
84
+ if (lang !== undefined)
85
+ return `"${lex}"@${lang.toLowerCase()}`;
86
+ if (dt === undefined || dt === XSD_STRING)
87
+ return `"${lex}"`; // plain / xsd:string
88
+ try {
89
+ if (INTEGER_TYPES.has(dt))
90
+ return canonIntegerTerm(lex) ?? verbatim(lex, dt);
91
+ if (dt === XSD + 'decimal')
92
+ return wrap(canonDecimal(lex), dt);
93
+ if (dt === XSD + 'boolean')
94
+ return wrap(canonBoolean(lex), dt);
95
+ if (dt === XSD + 'double')
96
+ return wrap(canonDouble(lex, false), dt);
97
+ if (dt === XSD + 'float')
98
+ return wrap(canonDouble(lex, true), dt);
99
+ if (dt === XSD + 'dateTime')
100
+ return wrap(canonDateTime(lex), dt);
101
+ if (dt === XSD + 'time')
102
+ return wrap(canonTime(lex), dt);
103
+ if (dt === XSD + 'date')
104
+ return wrap(canonDate(lex), dt);
105
+ if (dt === XSD + 'gYear')
106
+ return wrap(canonGYear(lex), dt);
107
+ if (dt === XSD + 'gYearMonth')
108
+ return wrap(canonGYearMonth(lex), dt);
109
+ if (dt === XSD + 'gMonthDay')
110
+ return wrap(canonGMonthDay(lex), dt);
111
+ if (dt === XSD + 'gMonth')
112
+ return wrap(canonGMonth(lex), dt);
113
+ if (dt === XSD + 'gDay')
114
+ return wrap(canonGDay(lex), dt);
115
+ if (DURATION_TYPES.has(dt))
116
+ return wrap(canonDuration(lex, dt), dt);
117
+ }
118
+ catch {
119
+ return verbatim(lex, dt); // invalid lexical → escaping-normalized, otherwise verbatim
120
+ }
121
+ return verbatim(lex, dt); // datatype the deployed store leaves verbatim
122
+ }
123
+ // Both helpers emit the bracketed N-Triples form; `verbatim` is `wrap` of the
124
+ // (escaping-normalized) lexical unchanged. Kept distinct for call-site intent.
125
+ const wrap = (canonLex, dt) => `"${canonLex}"^^<${dt}>`;
126
+ const verbatim = (lex, dt) => `"${lex}"^^<${dt}>`;
127
+ // ── literal content escaping ───────────────────────────────────────────────────
128
+ const ESCAPE_DECODE = /\\(u[0-9A-Fa-f]{4}|U[0-9A-Fa-f]{8}|[tbnrf"'\\])/g;
129
+ function normalizeEscaping(lex) {
130
+ const decoded = lex.replace(ESCAPE_DECODE, (whole, e) => {
131
+ const c = e[0];
132
+ if (c === 'u' || c === 'U') {
133
+ const cp = parseInt(e.slice(1), 16);
134
+ // A \U escape can encode up to 0xFFFFFFFF, but only ≤0x10FFFF is a valid
135
+ // Unicode scalar. String.fromCodePoint THROWS above that; oxigraph rejects
136
+ // the literal. Guard so canon never throws (it runs before the per-type
137
+ // try/catch and on EVERY literal) — leave the out-of-range escape undecoded.
138
+ if (cp > 0x10ffff)
139
+ return whole;
140
+ return String.fromCodePoint(cp);
141
+ }
142
+ switch (e) {
143
+ case 't': return '\t';
144
+ case 'b': return '\b';
145
+ case 'n': return '\n';
146
+ case 'r': return '\r';
147
+ case 'f': return '\f';
148
+ case '"': return '"';
149
+ case "'": return "'";
150
+ case '\\': return '\\';
151
+ default: return whole;
152
+ }
153
+ });
154
+ // re-emit oxigraph's minimal escaping (escapeNQuadsLiteral): \ " \n \r only.
155
+ return decoded.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\n/g, '\\n').replace(/\r/g, '\\r');
156
+ }
157
+ // Decode N-Triples UCHAR escapes (\uXXXX / \UXXXXXXXX) inside a datatype IRI, as
158
+ // oxigraph does on parse. Out-of-range \U (> U+10FFFF) is left undecoded (oxigraph
159
+ // would reject the literal; we must not throw).
160
+ function decodeIriEscapes(iri) {
161
+ if (!iri.includes('\\'))
162
+ return iri;
163
+ return iri.replace(/\\(u[0-9A-Fa-f]{4}|U[0-9A-Fa-f]{8})/g, (whole, e) => {
164
+ const cp = parseInt(e.slice(1), 16);
165
+ return cp > 0x10ffff ? whole : String.fromCodePoint(cp);
166
+ });
167
+ }
168
+ // oxigraph stores temporal values as seconds-since-0001-01-01 in the same i128/1e18
169
+ // Decimal as xsd:decimal/duration. A date/time whose scaled seconds overflow i128
170
+ // fails to parse and is kept VERBATIM, so a foldable timezone / T24 roll / fraction
171
+ // strip must NOT be applied to it. Replicated here via a proleptic-Gregorian day
172
+ // count. (Byte-exact vs oxigraph for dateTime/date; for the bare g-types the cliff
173
+ // may differ by ≤1 year at the ~5.39e12 boundary — a value impossible in real data.)
174
+ function daysFromCivil(y, m, d) {
175
+ const yy = m <= 2n ? y - 1n : y;
176
+ const era = (yy >= 0n ? yy : yy - 399n) / 400n;
177
+ const yoe = yy - era * 400n;
178
+ const doy = (153n * (m + (m > 2n ? -3n : 9n)) + 2n) / 5n + d - 1n;
179
+ const doe = yoe * 365n + yoe / 4n - yoe / 100n + doy;
180
+ return era * 146097n + doe - 719468n;
181
+ }
182
+ // Inverse of daysFromCivil: proleptic-Gregorian (y,m,d) from a signed day count
183
+ // (days since 1970-01-01). Standard Howard Hinnant algorithm. Used to roll the
184
+ // DATE when a timezone offset pushes a dateTime across midnight during the
185
+ // backend-independent UTC normalization (OT-RFC-57).
186
+ function civilFromDays(zIn) {
187
+ const z = zIn + 719468n;
188
+ const era = (z >= 0n ? z : z - 146096n) / 146097n;
189
+ const doe = z - era * 146097n; // [0, 146096]
190
+ const yoe = (doe - doe / 1460n + doe / 36524n - doe / 146096n) / 365n; // [0, 399]
191
+ const y = yoe + era * 400n;
192
+ const doy = doe - (365n * yoe + yoe / 4n - yoe / 100n); // [0, 365]
193
+ const mp = (5n * doy + 2n) / 153n; // [0, 11]
194
+ const d = doy - (153n * mp + 2n) / 5n + 1n; // [1, 31]
195
+ const m = mp < 10n ? mp + 3n : mp - 9n; // [1, 12]
196
+ return { y: m <= 2n ? y + 1n : y, m, d };
197
+ }
198
+ // OT-RFC-57: the UTC date of "midnight in the given tz" — the backend-independent
199
+ // form for xsd:date / gYear / gYearMonth. Blazegraph interprets the value at 00:00
200
+ // in its tz, converts to UTC, and takes the UTC date; a positive offset rolls the
201
+ // date back a day. offsetMin=0 (Z / no-tz) ⇒ the date is unchanged.
202
+ function utcDateFromMidnight(y, mo, d, offsetMin) {
203
+ const days = daysFromCivil(y, mo, d) + BigInt(Math.floor((0 - offsetMin) / 1440));
204
+ return civilFromDays(days);
205
+ }
206
+ function temporalInRange(yearStr, mo, dd, hh = 0, mi = 0, ss = 0) {
207
+ const seconds = (daysFromCivil(BigInt(yearStr), BigInt(mo), BigInt(dd)) + 719162n) * 86400n +
208
+ BigInt(hh) * 3600n + BigInt(mi) * 60n + BigInt(ss);
209
+ const scaled = seconds * DEC_SCALE;
210
+ return scaled >= I128_MIN && scaled <= I128_MAX;
211
+ }
212
+ // ── xsd:integer family ─────────────────────────────────────────────────────────
213
+ function canonIntegerTerm(lex) {
214
+ if (!/^[+-]?\d+$/.test(lex))
215
+ return null; // at most one sign; "+-1" etc. → verbatim
216
+ const v = BigInt(lex.replace(/^\+/, '')); // BigInt rejects a leading '+'
217
+ if (v < I64_MIN || v > I64_MAX)
218
+ return null; // outside i64 → verbatim (xsd:integer included)
219
+ return `"${v.toString()}"^^<${XSD_INTEGER}>`;
220
+ }
221
+ // ── xsd:boolean ────────────────────────────────────────────────────────────────
222
+ function canonBoolean(lex) {
223
+ if (lex === 'true' || lex === '1')
224
+ return 'true';
225
+ if (lex === 'false' || lex === '0')
226
+ return 'false';
227
+ throw new Error(`invalid xsd:boolean: ${lex}`);
228
+ }
229
+ // ── xsd:decimal ────────────────────────────────────────────────────────────────
230
+ function canonDecimal(lex) {
231
+ const m = /^([+-]?)(\d*)(?:\.(\d*))?$/.exec(lex);
232
+ if (!m || (m[2] === '' && (m[3] === undefined || m[3] === '')))
233
+ throw new Error(`invalid xsd:decimal: ${lex}`);
234
+ const intRaw = m[2].replace(/^0+/, '');
235
+ const frac = (m[3] ?? '').replace(/0+$/, '');
236
+ // oxigraph stores xsd:decimal as the SAME i128 / 10^18 fixed-point as duration
237
+ // seconds: a value needing more than 18 fractional digits, or whose 10^18-scaled
238
+ // magnitude overflows i128, fails to parse and is kept VERBATIM.
239
+ if (frac.length > 18)
240
+ throw new Error('xsd:decimal sub-1e-18');
241
+ const scaled = BigInt((intRaw || '0') + frac.padEnd(18, '0'));
242
+ const signed = m[1] === '-' ? -scaled : scaled;
243
+ if (signed < I128_MIN || signed > I128_MAX)
244
+ throw new Error('xsd:decimal overflow i128');
245
+ const int = intRaw === '' ? '0' : intRaw;
246
+ const sign = m[1] === '-' && !(int === '0' && frac === '') ? '-' : '';
247
+ return frac === '' ? `${sign}${int}` : `${sign}${int}.${frac}`;
248
+ }
249
+ // ── xsd:double / xsd:float ─────────────────────────────────────────────────────
250
+ function canonDouble(lex, isFloat) {
251
+ let n = parseXsdDouble(lex);
252
+ if (isFloat)
253
+ n = Math.fround(n);
254
+ if (Number.isNaN(n))
255
+ return 'NaN';
256
+ if (n === Infinity)
257
+ return 'INF';
258
+ if (n === -Infinity)
259
+ return '-INF';
260
+ // OT-RFC-57: negative zero folds to "0". Blazegraph drops the sign on write
261
+ // ("-0.0"^^double → stored "0.0" → value 0), while oxigraph keeps "-0"; emitting
262
+ // "0" for both signed zeros makes canon(input) == canon(store-readback) on either
263
+ // backend. (The IEEE-754 -0/+0 distinction is not consensus-observable here.)
264
+ if (n === 0)
265
+ return '0';
266
+ const neg = n < 0;
267
+ const a = Math.abs(n);
268
+ // double: V8's a.toString() IS the shortest round-trip; only ties need the
269
+ // away-from-zero correction. float: V8 has no f32-shortest, so search it.
270
+ const shortest = isFloat ? shortestFloat32String(a) : roundTiesAwayFromZero(a, a.toString(), false);
271
+ const plain = expandToPlainDecimal(shortest);
272
+ return neg ? `-${plain}` : plain;
273
+ }
274
+ // V8's Number→string breaks shortest-representation ties round-half-to-EVEN, but
275
+ // oxigraph (Rust) breaks them round-half-AWAY-from-zero. They diverge only when the
276
+ // value sits EXACTLY between two equal-length shortest decimals (e.g. the f64
277
+ // 738507753103385.25 → V8 ".2", Rust ".3"). Detect that tie with exact integer
278
+ // arithmetic and pick the away-from-zero neighbour to match oxigraph.
279
+ function roundTiesAwayFromZero(a, shortest, isFloat) {
280
+ const m = /^(\d+)(?:\.(\d+))?(?:[eE]([+-]?\d+))?$/.exec(shortest);
281
+ if (!m)
282
+ return shortest;
283
+ const digits = m[1] + (m[2] ?? '');
284
+ const D = BigInt(digits);
285
+ const E = (m[3] ? parseInt(m[3], 10) : 0) - (m[2] ? m[2].length : 0); // value = D × 10^E
286
+ const up = D + 1n; // away-from-zero neighbour at the same digit length (a ≥ 0)
287
+ const rt = (x) => (isFloat ? Math.fround(x) : x);
288
+ if (rt(Number(`${up}e${E}`)) !== a)
289
+ return shortest; // up-neighbour doesn't round-trip → no tie
290
+ // Exact tie test: 2·a == (2D+1) × 10^E, with a = num/den from the IEEE-754 bits.
291
+ const [num, den] = f64Fraction(a);
292
+ let lhs = 2n * num;
293
+ let rhs = (2n * D + 1n) * den;
294
+ if (E >= 0)
295
+ rhs *= 10n ** BigInt(E);
296
+ else
297
+ lhs *= 10n ** BigInt(-E);
298
+ return lhs === rhs ? `${up}e${E}` : shortest;
299
+ }
300
+ // Exact value of a finite |f64| as num/den (den a power of two) from its bits.
301
+ function f64Fraction(a) {
302
+ const dv = new DataView(new ArrayBuffer(8));
303
+ dv.setFloat64(0, a);
304
+ const bits = dv.getBigUint64(0);
305
+ const exp = Number((bits >> 52n) & 0x7ffn);
306
+ const fracBits = bits & 0xfffffffffffffn;
307
+ const mant = exp === 0 ? fracBits : fracBits | (1n << 52n);
308
+ const e = (exp === 0 ? -1074 : exp - 1075);
309
+ return e >= 0 ? [mant << BigInt(e), 1n] : [mant, 1n << BigInt(-e)];
310
+ }
311
+ function parseXsdDouble(lex) {
312
+ // oxigraph parses doubles with Rust's lenient f64::from_str: case-INSENSITIVE
313
+ // nan / inf / infinity (with an optional sign) all parse, not just the XSD
314
+ // spellings NaN / INF / -INF. Match it so e.g. "infinity"/"NaN"/"-inf" canon to
315
+ // the oxigraph forms INF / NaN / -INF instead of staying verbatim.
316
+ if (/^[+-]?nan$/i.test(lex))
317
+ return NaN;
318
+ if (/^\+?inf(inity)?$/i.test(lex))
319
+ return Infinity;
320
+ if (/^-inf(inity)?$/i.test(lex))
321
+ return -Infinity;
322
+ if (!/^[+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?$/.test(lex))
323
+ throw new Error(`invalid xsd:double: ${lex}`);
324
+ return Number(lex);
325
+ }
326
+ // Shortest decimal that round-trips to the f32 `a`, matching Rust's f32 formatting.
327
+ // V8 has no native f32-shortest, and a.toPrecision(p)/toExponential round `a` to
328
+ // NEAREST — which can miss the round-tripping p-digit decimal sitting on the other
329
+ // side of `a` (a's f32 rounding interval is wider than its f64 one). So at each
330
+ // precision we test the nearest p-digit mantissa AND its ±1 neighbours (exact
331
+ // integers, no float-grid error), keep those whose f32 round-trip equals a, and
332
+ // pick the closest to a — ties to the away-from-zero (larger) value, as Rust does.
333
+ function shortestFloat32String(a) {
334
+ for (let p = 1; p <= 9; p++) {
335
+ const m = /^(\d)(?:\.(\d+))?e([+-]\d+)$/.exec(a.toExponential(p - 1));
336
+ if (!m)
337
+ break;
338
+ const mant = m[1] + (m[2] ?? ''); // p significant digits
339
+ const e10 = parseInt(m[3], 10) - (p - 1); // value = mant × 10^e10
340
+ const base = BigInt(mant);
341
+ const valid = [];
342
+ for (const v of [base, base - 1n, base + 1n]) {
343
+ if (v <= 0n)
344
+ continue;
345
+ const c = Number(`${v}e${e10}`);
346
+ if (Math.fround(c) === a)
347
+ valid.push(c);
348
+ }
349
+ if (valid.length) {
350
+ valid.sort((x, y) => Math.abs(x - a) - Math.abs(y - a) || y - x); // closest; tie → away from zero
351
+ return valid[0].toString();
352
+ }
353
+ }
354
+ return a.toString();
355
+ }
356
+ function expandToPlainDecimal(s) {
357
+ const m = /^(\d+)(?:\.(\d+))?[eE]([+-]?\d+)$/.exec(s);
358
+ if (!m)
359
+ return s;
360
+ const intPart = m[1];
361
+ const frac = m[2] ?? '';
362
+ const exp = parseInt(m[3], 10);
363
+ const digits = intPart + frac;
364
+ const pointPos = intPart.length + exp;
365
+ if (pointPos <= 0)
366
+ return stripTrailingZeros(`0.${'0'.repeat(-pointPos)}${digits}`);
367
+ if (pointPos >= digits.length)
368
+ return digits + '0'.repeat(pointPos - digits.length);
369
+ return stripTrailingZeros(`${digits.slice(0, pointPos)}.${digits.slice(pointPos)}`);
370
+ }
371
+ function stripTrailingZeros(s) {
372
+ if (!s.includes('.'))
373
+ return s;
374
+ return s.replace(/0+$/, '').replace(/\.$/, '');
375
+ }
376
+ // ── date/time family ───────────────────────────────────────────────────────────
377
+ // Returns the offset MAGNITUDE in minutes (signed) for the
378
+ // backend-independent UTC normalization of xsd:dateTime/xsd:time (OT-RFC-57).
379
+ // hadTz=false ⇒ no timezone present (a bare dateTime is normalized to UTC and
380
+ // gains a Z, matching Blazegraph/Neptune). Malformed/out-of-range tz → throw
381
+ // (→ the literal is kept verbatim, as oxigraph does).
382
+ function splitTzToOffset(s) {
383
+ const m = /(Z|[+-]\d{2}:\d{2})$/.exec(s);
384
+ if (!m)
385
+ return { body: s, offsetMin: 0, hadTz: false };
386
+ const tz = m[1];
387
+ const body = s.slice(0, s.length - tz.length);
388
+ if (tz === 'Z')
389
+ return { body, offsetMin: 0, hadTz: true };
390
+ const h = parseInt(tz.slice(1, 3), 10);
391
+ const mi = parseInt(tz.slice(4, 6), 10);
392
+ if (mi > 59 || h * 60 + mi > 840)
393
+ throw new Error(`invalid tz: ${tz}`);
394
+ const mag = h * 60 + mi;
395
+ return { body, offsetMin: tz[0] === '-' ? -mag : mag, hadTz: true };
396
+ }
397
+ // Normalize a fractional-seconds group ('.ddd' or undefined): TRUNCATE to at most
398
+ // 3 digits (milliseconds — the backend-independent precision floor; a lossy store
399
+ // such as Blazegraph keeps only ms), then strip trailing zeros; drop entirely if
400
+ // empty. Truncate, NOT round (matches Blazegraph). (OT-RFC-57)
401
+ function normFrac(frac) {
402
+ if (frac === undefined)
403
+ return '';
404
+ const d = frac.slice(1, 4).replace(/0+$/, ''); // at most 3 digits, then strip trailing zeros
405
+ return d === '' ? '' : `.${d}`;
406
+ }
407
+ // Proleptic-Gregorian leap test on the astronomical year number (handles negative
408
+ // and arbitrarily large years via BigInt).
409
+ function isLeapYear(yearStr) {
410
+ const y = BigInt(yearStr);
411
+ return (y % 4n === 0n && y % 100n !== 0n) || y % 400n === 0n;
412
+ }
413
+ function daysInMonth(yearStr, mo) {
414
+ return [31, isLeapYear(yearStr) ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31][mo - 1];
415
+ }
416
+ // Day after (yearStr, mo, dd). Year crosses are computed on the signed numeric
417
+ // year (BigInt) then re-emitted min-4-digit, sign-preserved (…→0000, 9999→10000).
418
+ function rollNextDay(yearStr, mo, dd) {
419
+ let ny = BigInt(yearStr);
420
+ let nmo = mo;
421
+ let nd = dd + 1;
422
+ if (nd > daysInMonth(yearStr, mo)) {
423
+ nd = 1;
424
+ nmo += 1;
425
+ if (nmo > 12) {
426
+ nmo = 1;
427
+ ny += 1n;
428
+ }
429
+ }
430
+ return `${fmtYear(ny)}-${pad2(nmo)}-${pad2(nd)}`;
431
+ }
432
+ function fmtYear(y) {
433
+ const neg = y < 0n;
434
+ const abs = (neg ? -y : y).toString().padStart(4, '0');
435
+ return neg ? `-${abs}` : abs;
436
+ }
437
+ const pad2 = (n) => String(n).padStart(2, '0');
438
+ // hh∈[0,24], mm∈[0,59], ss∈[0,59]; hour 24 is valid ONLY when it can roll, which
439
+ // oxigraph allows iff minute==0 OR the seconds value (incl. fraction) is 0.
440
+ // Returns whether the time rolls to the next day (hour 24 → 00). Throws on any
441
+ // out-of-range field or a non-rollable hour-24.
442
+ function validateClock(hh, mi, ss, fracNorm) {
443
+ if (hh > 24 || mi > 59 || ss > 59)
444
+ throw new Error('clock out of range');
445
+ if (hh === 24) {
446
+ const secZero = ss === 0 && fracNorm === '';
447
+ if (mi !== 0 && !secZero)
448
+ throw new Error('invalid hour-24');
449
+ return { rolls: true };
450
+ }
451
+ return { rolls: false };
452
+ }
453
+ // OT-RFC-57: the backend-independent value canon accepts any 4+-digit year (any
454
+ // number of leading zeros) and normalizes it via BigInt+fmtYear (min-4-digit, no
455
+ // leading zero). This matches Blazegraph, which on write STRIPS a leading-zero
456
+ // year to its value ("02026"^^gYear → "2026") — oxigraph instead keeps the invalid
457
+ // literal verbatim, but the CONVERGENCE oracle holds either way since canon(input)
458
+ // and canon(store-readback) both fold to the same value form (OT-RFC-57 §7.5).
459
+ const YEAR = '-?\\d{4,}';
460
+ // OT-RFC-57 backend-independent form: normalize to UTC (subtract the tz offset,
461
+ // rolling the DATE across midnight), truncate fraction to ms, always emit Z. A
462
+ // no-timezone dateTime is treated as UTC and gains a Z (matching Blazegraph /
463
+ // Neptune). This is the value-space form the publisher's input AND every
464
+ // backend's read-back converge to.
465
+ function canonDateTime(lex) {
466
+ const { body, offsetMin } = splitTzToOffset(lex);
467
+ const m = new RegExp(`^(${YEAR})-(\\d{2})-(\\d{2})T(\\d{2}):(\\d{2}):(\\d{2})(\\.\\d+)?$`).exec(body);
468
+ if (!m)
469
+ throw new Error('invalid xsd:dateTime');
470
+ const [, yy, mo, dd, hh, mi, ss, frac] = m;
471
+ const moN = +mo;
472
+ const ddN = +dd;
473
+ if (moN < 1 || moN > 12)
474
+ throw new Error('month');
475
+ if (ddN < 1 || ddN > daysInMonth(yy, moN))
476
+ throw new Error('day');
477
+ const fracNorm = normFrac(frac);
478
+ const { rolls } = validateClock(+hh, +mi, +ss, fracNorm);
479
+ // Base date as a day count; a T24:00 clock rolls one day and resets the hour to 0.
480
+ let days = daysFromCivil(BigInt(yy), BigInt(moN), BigInt(ddN));
481
+ const hourN = rolls ? 0 : +hh;
482
+ if (rolls)
483
+ days += 1n;
484
+ // UTC: subtract the offset (whole minutes); roll the date across midnight.
485
+ const totalMin = hourN * 60 + +mi - offsetMin;
486
+ days += BigInt(Math.floor(totalMin / 1440));
487
+ const minInDay = ((totalMin % 1440) + 1440) % 1440;
488
+ const { y, m: mm, d } = civilFromDays(days);
489
+ // Range-check the NORMALIZED UTC instant, not the lexical components: a tz offset
490
+ // or T24 roll can push a boundary value outside the i128 seconds range it would
491
+ // otherwise pass, emitting a leaf for a value the store can't represent stably
492
+ // (otReviewAgent). Out of range → verbatim (throw, caught upstream).
493
+ if (!temporalInRange(y.toString(), Number(mm), Number(d), Math.floor(minInDay / 60), minInDay % 60, +ss))
494
+ throw new Error('normalized dateTime overflows i128 seconds');
495
+ return `${fmtYear(y)}-${pad2(Number(mm))}-${pad2(Number(d))}T${pad2(Math.floor(minInDay / 60))}:${pad2(minInDay % 60)}:${ss}${fracNorm}Z`;
496
+ }
497
+ // OT-RFC-57: time has no date, so a tz offset just wraps the wall clock mod 24h;
498
+ // normalize to UTC + Z, ms-truncated.
499
+ function canonTime(lex) {
500
+ const { body, offsetMin } = splitTzToOffset(lex);
501
+ const m = /^(\d{2}):(\d{2}):(\d{2})(\.\d+)?$/.exec(body);
502
+ if (!m)
503
+ throw new Error('invalid xsd:time');
504
+ const [, hh, mi, ss, frac] = m;
505
+ const fracNorm = normFrac(frac);
506
+ const { rolls } = validateClock(+hh, +mi, +ss, fracNorm);
507
+ const hourN = rolls ? 0 : +hh;
508
+ const minInDay = (((hourN * 60 + +mi - offsetMin) % 1440) + 1440) % 1440;
509
+ return `${pad2(Math.floor(minInDay / 60))}:${pad2(minInDay % 60)}:${ss}${fracNorm}Z`;
510
+ }
511
+ // OT-RFC-57: xsd:date / gYear / gYearMonth normalize to the UTC date of
512
+ // midnight-in-tz, with NO timezone emitted (Blazegraph's value form).
513
+ function canonDate(lex) {
514
+ const { body, offsetMin } = splitTzToOffset(lex);
515
+ const m = new RegExp(`^(${YEAR})-(\\d{2})-(\\d{2})$`).exec(body);
516
+ if (!m)
517
+ throw new Error('invalid xsd:date');
518
+ const moN = +m[2];
519
+ const ddN = +m[3];
520
+ if (moN < 1 || moN > 12)
521
+ throw new Error('month');
522
+ if (ddN < 1 || ddN > daysInMonth(m[1], moN))
523
+ throw new Error('day');
524
+ const { y, m: mm, d } = utcDateFromMidnight(BigInt(m[1]), BigInt(moN), BigInt(ddN), offsetMin);
525
+ // Validate the NORMALIZED date (the tz roll can cross the year boundary) — see canonDateTime.
526
+ if (!temporalInRange(y.toString(), Number(mm), Number(d)))
527
+ throw new Error('normalized date overflows i128 seconds');
528
+ return `${fmtYear(y)}-${pad2(Number(mm))}-${pad2(Number(d))}`;
529
+ }
530
+ function canonGYear(lex) {
531
+ const { body, offsetMin } = splitTzToOffset(lex);
532
+ if (!new RegExp(`^${YEAR}$`).test(body))
533
+ throw new Error('invalid xsd:gYear');
534
+ const { y, m: mm, d } = utcDateFromMidnight(BigInt(body), 1n, 1n, offsetMin);
535
+ // Validate the NORMALIZED date (a negative offset can roll 01-01 into the prior year).
536
+ if (!temporalInRange(y.toString(), Number(mm), Number(d)))
537
+ throw new Error('normalized gYear overflows i128 seconds');
538
+ return fmtYear(y);
539
+ }
540
+ function canonGYearMonth(lex) {
541
+ const { body, offsetMin } = splitTzToOffset(lex);
542
+ const m = new RegExp(`^(${YEAR})-(\\d{2})$`).exec(body);
543
+ if (!m || +m[2] < 1 || +m[2] > 12)
544
+ throw new Error('invalid xsd:gYearMonth');
545
+ const { y, m: mm, d } = utcDateFromMidnight(BigInt(m[1]), BigInt(+m[2]), 1n, offsetMin);
546
+ // Validate the NORMALIZED date (the tz roll can cross the year boundary).
547
+ if (!temporalInRange(y.toString(), Number(mm), Number(d)))
548
+ throw new Error('normalized gYearMonth overflows i128 seconds');
549
+ return `${fmtYear(y)}-${pad2(Number(mm))}`;
550
+ }
551
+ // gMonthDay day bounds. oxigraph 0.5.5 validates --MM-DD against a NON-leap
552
+ // reference year, so --02-29 is rejected (kept verbatim) — February's max is 28
553
+ // here, unlike a real leap date which needs the year context of xsd:date.
554
+ const MONTH_MAX_DAY = [31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
555
+ // OT-RFC-57: gMonthDay / gMonth / gDay have no year/date context to convert a
556
+ // timezone into UTC. We therefore fold ONLY a UTC-equivalent zone (Z / +00:00 /
557
+ // -00:00 → offsetMin 0) to the no-timezone value form. A NON-UTC offset is kept
558
+ // VERBATIM (the whole literal, offset included): stripping it would silently
559
+ // COLLAPSE distinct values — "--06-29+14:00" and "--06-29-14:00" are different
560
+ // literals — onto one leaf (otReviewAgent). Verbatim keeps them distinct and defers
561
+ // to the store's own preservation; such exotic offsets on bare gregorian types are
562
+ // vanishingly rare and out of the consensus-verified set (see OT-RFC-57 §7.8).
563
+ function bareGregorian(lex, re, validate) {
564
+ const { body, offsetMin } = splitTzToOffset(lex);
565
+ const m = re.exec(body);
566
+ if (!m || !validate(m))
567
+ throw new Error('invalid bare gregorian');
568
+ return offsetMin === 0 ? body : lex; // fold UTC-equivalent zone only; else verbatim
569
+ }
570
+ function canonGMonthDay(lex) {
571
+ return bareGregorian(lex, /^--(\d{2})-(\d{2})$/, (m) => {
572
+ const moN = +m[1];
573
+ const ddN = +m[2];
574
+ return moN >= 1 && moN <= 12 && ddN >= 1 && ddN <= MONTH_MAX_DAY[moN - 1];
575
+ });
576
+ }
577
+ function canonGMonth(lex) {
578
+ return bareGregorian(lex, /^--(\d{2})$/, (m) => +m[1] >= 1 && +m[1] <= 12);
579
+ }
580
+ function canonGDay(lex) {
581
+ return bareGregorian(lex, /^---(\d{2})$/, (m) => +m[1] >= 1 && +m[1] <= 31);
582
+ }
583
+ // ── xsd:duration / dayTimeDuration / yearMonthDuration ─────────────────────────
584
+ // Value-space canonicalization mirroring oxsdatatypes Duration { months: i64,
585
+ // seconds: Decimal(i128 / 10^18) }. Parse to (months, scaledSeconds), reject if
586
+ // either overflows its integer type (→ verbatim), then re-emit the canonical
587
+ // component breakdown (Y=months/12, M=months%12; D/H/M/S from seconds).
588
+ // Seconds accept a trailing dot with no fraction (oxigraph: "PT1.S" → "PT1S") and a
589
+ // leading dot ("PT.5S" → "PT0.5S"), matching Rust's lenient parse.
590
+ const RE_DURATION = /^(-?)P(?:(\d+)Y)?(?:(\d+)M)?(?:(\d+)D)?(?:T(?:(\d+)H)?(?:(\d+)M)?(?:(\d+(?:\.\d*)?|\.\d+)S)?)?$/;
591
+ function canonDuration(lex, dt) {
592
+ const m = RE_DURATION.exec(lex);
593
+ if (!m)
594
+ throw new Error(`invalid duration: ${lex}`);
595
+ const [, sign, y, mo, d, h, mi, sTok] = m;
596
+ if (!y && !mo && !d && !h && !mi && !sTok)
597
+ throw new Error('empty duration'); // "P" / "PT"
598
+ const isYM = dt === XSD + 'yearMonthDuration';
599
+ const isDT = dt === XSD + 'dayTimeDuration';
600
+ // Subtype component constraints (oxigraph keeps a mis-componented subtype verbatim).
601
+ if (isYM && (d || h || mi || sTok))
602
+ throw new Error('yearMonthDuration: time/day component');
603
+ if (isDT && (y || mo))
604
+ throw new Error('dayTimeDuration: year/month component');
605
+ // Seconds token → whole + fractional (≤18 significant fractional digits; the
606
+ // oxsdatatypes Decimal scale is 10^18, so a 19th significant digit is rejected).
607
+ let sWhole = '0';
608
+ let fracScaled = 0n;
609
+ if (sTok) {
610
+ const dot = sTok.indexOf('.');
611
+ if (dot === -1) {
612
+ sWhole = sTok;
613
+ }
614
+ else {
615
+ sWhole = sTok.slice(0, dot) || '0';
616
+ const fracDigits = sTok.slice(dot + 1).replace(/0+$/, '');
617
+ if (fracDigits.length > 18)
618
+ throw new Error('sub-1e-18 seconds');
619
+ fracScaled = fracDigits === '' ? 0n : BigInt(fracDigits.padEnd(18, '0'));
620
+ }
621
+ }
622
+ const months = 12n * BigInt(y || '0') + BigInt(mo || '0');
623
+ const wholeSeconds = 86400n * BigInt(d || '0') + 3600n * BigInt(h || '0') + 60n * BigInt(mi || '0') + BigInt(sWhole);
624
+ const scaledSeconds = wholeSeconds * DEC_SCALE + fracScaled;
625
+ // Overflow at oxigraph's stored-integer boundaries (signed) → verbatim.
626
+ const neg = sign === '-';
627
+ const signedMonths = neg ? -months : months;
628
+ const signedScaled = neg ? -scaledSeconds : scaledSeconds;
629
+ if (signedMonths < I64_MIN || signedMonths > I64_MAX)
630
+ throw new Error('months overflow i64');
631
+ if (signedScaled < I128_MIN || signedScaled > I128_MAX)
632
+ throw new Error('seconds overflow i128');
633
+ // Re-derive canonical components from magnitudes (sign emitted once).
634
+ const yy = months / 12n;
635
+ const MM = months % 12n;
636
+ const totalWhole = scaledSeconds / DEC_SCALE;
637
+ const fracRem = scaledSeconds % DEC_SCALE;
638
+ const D = totalWhole / 86400n;
639
+ let rem = totalWhole % 86400n;
640
+ const H = rem / 3600n;
641
+ rem %= 3600n;
642
+ const Min = rem / 60n;
643
+ const S = rem % 60n;
644
+ let date = '';
645
+ if (yy > 0n)
646
+ date += `${yy}Y`;
647
+ if (MM > 0n)
648
+ date += `${MM}M`;
649
+ if (D > 0n)
650
+ date += `${D}D`;
651
+ let time = '';
652
+ if (H > 0n)
653
+ time += `${H}H`;
654
+ if (Min > 0n)
655
+ time += `${Min}M`;
656
+ if (S > 0n || fracRem > 0n) {
657
+ const fracStr = fracRem === 0n ? '' : `.${fracRem.toString().padStart(18, '0').replace(/0+$/, '')}`;
658
+ time += `${S}${fracStr}S`;
659
+ }
660
+ const body = time ? `${date}T${time}` : date;
661
+ // All-zero canonical form is subtype-dependent: yearMonthDuration → "P0M",
662
+ // duration / dayTimeDuration → "PT0S".
663
+ if (body === '')
664
+ return isYM ? 'P0M' : 'PT0S';
665
+ return `${neg ? '-' : ''}P${body}`;
666
+ }
667
+ //# sourceMappingURL=term-canon.js.map