@origintrail-official/dkg-core 10.0.0 → 10.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/crypto/canonicalize.d.ts.map +1 -1
- package/dist/crypto/canonicalize.js +6 -1
- package/dist/crypto/canonicalize.js.map +1 -1
- package/dist/crypto/index.d.ts +1 -0
- package/dist/crypto/index.d.ts.map +1 -1
- package/dist/crypto/index.js +1 -0
- package/dist/crypto/index.js.map +1 -1
- package/dist/crypto/term-canon.d.ts +2 -0
- package/dist/crypto/term-canon.d.ts.map +1 -0
- package/dist/crypto/term-canon.js +667 -0
- package/dist/crypto/term-canon.js.map +1 -0
- package/dist/ensure-dkg-node-config.d.ts +27 -0
- package/dist/ensure-dkg-node-config.d.ts.map +1 -1
- package/dist/ensure-dkg-node-config.js +55 -2
- package/dist/ensure-dkg-node-config.js.map +1 -1
- package/dist/errors.d.ts +18 -0
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +21 -0
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +7 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -2
- package/dist/index.js.map +1 -1
- package/dist/log-redaction.d.ts +45 -0
- package/dist/log-redaction.d.ts.map +1 -0
- package/dist/log-redaction.js +163 -0
- package/dist/log-redaction.js.map +1 -0
- package/dist/logger.d.ts +18 -2
- package/dist/logger.d.ts.map +1 -1
- package/dist/logger.js +23 -4
- package/dist/logger.js.map +1 -1
- package/dist/proto/index.d.ts +1 -1
- package/dist/proto/index.d.ts.map +1 -1
- package/dist/proto/index.js +1 -1
- package/dist/proto/index.js.map +1 -1
- package/dist/proto/storage-ack.d.ts +9 -0
- package/dist/proto/storage-ack.d.ts.map +1 -1
- package/dist/proto/storage-ack.js +13 -0
- package/dist/proto/storage-ack.js.map +1 -1
- package/dist/protocol-router.d.ts +1 -0
- package/dist/protocol-router.d.ts.map +1 -1
- package/dist/protocol-router.js +20 -0
- package/dist/protocol-router.js.map +1 -1
- package/dist/publisher-extension.d.ts +0 -12
- package/dist/publisher-extension.d.ts.map +1 -1
- package/dist/publisher-extension.js +7 -7
- package/dist/publisher-extension.js.map +1 -1
- package/dist/rdf-literal-size.d.ts +49 -0
- package/dist/rdf-literal-size.d.ts.map +1 -0
- package/dist/rdf-literal-size.js +111 -0
- package/dist/rdf-literal-size.js.map +1 -0
- package/dist/setup-network.d.ts +64 -0
- package/dist/setup-network.d.ts.map +1 -0
- package/dist/setup-network.js +58 -0
- package/dist/setup-network.js.map +1 -0
- package/dist/telemetry-api.d.ts +89 -0
- package/dist/telemetry-api.d.ts.map +1 -0
- package/dist/telemetry-api.js +124 -0
- package/dist/telemetry-api.js.map +1 -0
- package/package.json +4 -1
|
@@ -0,0 +1,667 @@
|
|
|
1
|
+
// Protocol-defined, BACKEND-INDEPENDENT canonicalization of an RDF object term,
|
|
2
|
+
// applied at the V10 merkle leaf (tripleContentV10) so EVERY node computes the
|
|
3
|
+
// identical leaf for the same triple regardless of which triple store it runs
|
|
4
|
+
// (oxigraph, blazegraph, a SPARQL endpoint, future backends) and which version.
|
|
5
|
+
// Without this the leaf delegated literal canonicalization to whatever string the
|
|
6
|
+
// backend emitted, so a publisher sealing pre-store and a peer recomputing
|
|
7
|
+
// post-store could hash different serializations of the SAME triple →
|
|
8
|
+
// MERKLE_MISMATCH_IN_SWM (okf-dkg-vm-validation-report.md), and two nodes on
|
|
9
|
+
// different backends could fork RandomSampling (the contract hashes this exact
|
|
10
|
+
// content: leaf = keccak256(content)).
|
|
11
|
+
//
|
|
12
|
+
// The canonical form is DEFINED to equal the value-space canonicalization the
|
|
13
|
+
// network already deploys (oxigraph 0.5.5), verified byte-for-byte by the
|
|
14
|
+
// oxigraph-oracle test (packages/publisher/test/term-canon-oracle.test.ts), so it
|
|
15
|
+
// is the IDENTITY on already-canonical (store-loaded) terms ⇒ no migration; a
|
|
16
|
+
// coordinated release suffices.
|
|
17
|
+
//
|
|
18
|
+
// STRUCTURE (per consensus primitive: easy to audit, hard to drift):
|
|
19
|
+
// parse → validate against oxigraph's EXACT accepted set → canonicalize if
|
|
20
|
+
// valid, else return the escaping-normalized term VERBATIM. Every normalize is
|
|
21
|
+
// gated by a validate: oxigraph keeps an ill-typed-but-syntactically-odd literal
|
|
22
|
+
// verbatim (e.g. "not-a-date+00:00"^^xsd:dateTime), so we must NOT mutate it.
|
|
23
|
+
//
|
|
24
|
+
// Covered (all verified against oxigraph 0.5.5 — see the oracle test):
|
|
25
|
+
// - literal-content ESCAPING (decode N-Triples escapes, re-emit oxigraph's
|
|
26
|
+
// minimal \ " \n \r escaping); BARE datatype IRIs ("v"^^IRI w/o <>) folded to
|
|
27
|
+
// the bracketed form; language-tag lowercasing; xsd:string elision.
|
|
28
|
+
// - xsd:integer family (collapse→xsd:integer iff value∈i64; xsd:integer
|
|
29
|
+
// arbitrary precision); xsd:decimal; xsd:boolean; xsd:double / xsd:float.
|
|
30
|
+
// - the date/time family with FULL value-space validation + the +00:00/-00:00→Z
|
|
31
|
+
// tz fold (xsd:dateTime/time/date/gYear/gYearMonth/gMonthDay/gMonth/gDay) and
|
|
32
|
+
// the T24:00→next-day roll (oxigraph rolls iff minute==0 OR seconds==0).
|
|
33
|
+
// - xsd:duration / dayTimeDuration / yearMonthDuration: FULL value-space
|
|
34
|
+
// normalization — leading-zero strip + component carry/overflow (months as
|
|
35
|
+
// i64, seconds as a 10^18-scaled i128 fixed-point, mirroring oxsdatatypes),
|
|
36
|
+
// subtype-component constraints, all-zero → PT0S / P0M.
|
|
37
|
+
// Other datatypes (hexBinary, base64Binary, anyURI, token, custom IRIs, …) are
|
|
38
|
+
// returned with normalized escaping but otherwise verbatim — matching oxigraph.
|
|
39
|
+
const XSD = 'http://www.w3.org/2001/XMLSchema#';
|
|
40
|
+
const XSD_STRING = XSD + 'string';
|
|
41
|
+
const XSD_INTEGER = XSD + 'integer';
|
|
42
|
+
const INTEGER_TYPES = new Set([
|
|
43
|
+
'integer', 'int', 'long', 'short', 'byte',
|
|
44
|
+
'nonNegativeInteger', 'positiveInteger', 'nonPositiveInteger', 'negativeInteger',
|
|
45
|
+
'unsignedLong', 'unsignedInt', 'unsignedShort', 'unsignedByte',
|
|
46
|
+
].map((t) => XSD + t));
|
|
47
|
+
const DURATION_TYPES = new Set(['duration', 'dayTimeDuration', 'yearMonthDuration'].map((t) => XSD + t));
|
|
48
|
+
// oxigraph parses integers (incl. xsd:integer) into a signed 64-bit int and only
|
|
49
|
+
// THEN re-serializes canonically; a value outside i64 fails to parse and is kept
|
|
50
|
+
// VERBATIM (sign + leading zeros preserved) — for EVERY integer type, xsd:integer
|
|
51
|
+
// included. So canonicalization (collapse-to-integer + strip sign/zeros) applies
|
|
52
|
+
// iff the value fits i64.
|
|
53
|
+
const I64_MIN = -9223372036854775808n;
|
|
54
|
+
const I64_MAX = 9223372036854775807n;
|
|
55
|
+
// oxsdatatypes Duration: months is i64, seconds is a Decimal == i128 scaled by
|
|
56
|
+
// 10^18. A duration whose normalized months/seconds overflow these is rejected by
|
|
57
|
+
// oxigraph (kept verbatim), so we must reject (→ verbatim) at the same boundary.
|
|
58
|
+
const I128_MIN = -(1n << 127n);
|
|
59
|
+
const I128_MAX = (1n << 127n) - 1n;
|
|
60
|
+
const DEC_SCALE = 10n ** 18n;
|
|
61
|
+
// A literal is "<lex>"(@tag | ^^<IRI> | ^^IRI). N-Triples mandates the bracketed
|
|
62
|
+
// datatype, but oxigraph's storage adapter also accepts the BARE "v"^^IRI form on
|
|
63
|
+
// input and canonicalizes it to "v"^^<IRI>, so we accept (m[3] bracketed, m[4]
|
|
64
|
+
// bare) and always re-emit bracketed. The bare alternative is `[^<]…` so a
|
|
65
|
+
// MALFORMED bracketed datatype (e.g. "a"^^<b>extra, "x"^^<>) does NOT get
|
|
66
|
+
// mis-captured as a bare IRI — it fails the match and is returned verbatim, as
|
|
67
|
+
// oxigraph does.
|
|
68
|
+
const RE_LITERAL = /^"((?:[^"\\]|\\.)*)"(?:@([A-Za-z0-9-]+)|\^\^(?:<([^>]+)>|([^<].*)))?$/;
|
|
69
|
+
export function canonicalizeObjectTermForHash(object) {
|
|
70
|
+
if (object.length === 0 || object.charCodeAt(0) !== 34 /* " */)
|
|
71
|
+
return object; // IRI / blank / genid
|
|
72
|
+
const m = RE_LITERAL.exec(object);
|
|
73
|
+
if (!m)
|
|
74
|
+
return object;
|
|
75
|
+
const lang = m[2];
|
|
76
|
+
// oxigraph decodes N-Triples UCHAR (\uXXXX / \UXXXXXXXX) escapes inside the
|
|
77
|
+
// datatype IRI on parse, so the canonical form (and any datatype matching below)
|
|
78
|
+
// must run on the decoded IRI — e.g. <…XMLSchema#integer> ≡ xsd:integer.
|
|
79
|
+
const dtRaw = m[3] ?? m[4];
|
|
80
|
+
const dt = dtRaw === undefined ? undefined : decodeIriEscapes(dtRaw);
|
|
81
|
+
// Literal CONTENT escaping is normalized for every literal (a store decodes
|
|
82
|
+
// \uXXXX / \t / \U… to raw UTF-8 and re-emits only \ " \n \r escaped).
|
|
83
|
+
const lex = normalizeEscaping(m[1]);
|
|
84
|
+
if (lang !== undefined)
|
|
85
|
+
return `"${lex}"@${lang.toLowerCase()}`;
|
|
86
|
+
if (dt === undefined || dt === XSD_STRING)
|
|
87
|
+
return `"${lex}"`; // plain / xsd:string
|
|
88
|
+
try {
|
|
89
|
+
if (INTEGER_TYPES.has(dt))
|
|
90
|
+
return canonIntegerTerm(lex) ?? verbatim(lex, dt);
|
|
91
|
+
if (dt === XSD + 'decimal')
|
|
92
|
+
return wrap(canonDecimal(lex), dt);
|
|
93
|
+
if (dt === XSD + 'boolean')
|
|
94
|
+
return wrap(canonBoolean(lex), dt);
|
|
95
|
+
if (dt === XSD + 'double')
|
|
96
|
+
return wrap(canonDouble(lex, false), dt);
|
|
97
|
+
if (dt === XSD + 'float')
|
|
98
|
+
return wrap(canonDouble(lex, true), dt);
|
|
99
|
+
if (dt === XSD + 'dateTime')
|
|
100
|
+
return wrap(canonDateTime(lex), dt);
|
|
101
|
+
if (dt === XSD + 'time')
|
|
102
|
+
return wrap(canonTime(lex), dt);
|
|
103
|
+
if (dt === XSD + 'date')
|
|
104
|
+
return wrap(canonDate(lex), dt);
|
|
105
|
+
if (dt === XSD + 'gYear')
|
|
106
|
+
return wrap(canonGYear(lex), dt);
|
|
107
|
+
if (dt === XSD + 'gYearMonth')
|
|
108
|
+
return wrap(canonGYearMonth(lex), dt);
|
|
109
|
+
if (dt === XSD + 'gMonthDay')
|
|
110
|
+
return wrap(canonGMonthDay(lex), dt);
|
|
111
|
+
if (dt === XSD + 'gMonth')
|
|
112
|
+
return wrap(canonGMonth(lex), dt);
|
|
113
|
+
if (dt === XSD + 'gDay')
|
|
114
|
+
return wrap(canonGDay(lex), dt);
|
|
115
|
+
if (DURATION_TYPES.has(dt))
|
|
116
|
+
return wrap(canonDuration(lex, dt), dt);
|
|
117
|
+
}
|
|
118
|
+
catch {
|
|
119
|
+
return verbatim(lex, dt); // invalid lexical → escaping-normalized, otherwise verbatim
|
|
120
|
+
}
|
|
121
|
+
return verbatim(lex, dt); // datatype the deployed store leaves verbatim
|
|
122
|
+
}
|
|
123
|
+
// Both helpers emit the bracketed N-Triples form; `verbatim` is `wrap` of the
|
|
124
|
+
// (escaping-normalized) lexical unchanged. Kept distinct for call-site intent.
|
|
125
|
+
const wrap = (canonLex, dt) => `"${canonLex}"^^<${dt}>`;
|
|
126
|
+
const verbatim = (lex, dt) => `"${lex}"^^<${dt}>`;
|
|
127
|
+
// ── literal content escaping ───────────────────────────────────────────────────
|
|
128
|
+
const ESCAPE_DECODE = /\\(u[0-9A-Fa-f]{4}|U[0-9A-Fa-f]{8}|[tbnrf"'\\])/g;
|
|
129
|
+
function normalizeEscaping(lex) {
|
|
130
|
+
const decoded = lex.replace(ESCAPE_DECODE, (whole, e) => {
|
|
131
|
+
const c = e[0];
|
|
132
|
+
if (c === 'u' || c === 'U') {
|
|
133
|
+
const cp = parseInt(e.slice(1), 16);
|
|
134
|
+
// A \U escape can encode up to 0xFFFFFFFF, but only ≤0x10FFFF is a valid
|
|
135
|
+
// Unicode scalar. String.fromCodePoint THROWS above that; oxigraph rejects
|
|
136
|
+
// the literal. Guard so canon never throws (it runs before the per-type
|
|
137
|
+
// try/catch and on EVERY literal) — leave the out-of-range escape undecoded.
|
|
138
|
+
if (cp > 0x10ffff)
|
|
139
|
+
return whole;
|
|
140
|
+
return String.fromCodePoint(cp);
|
|
141
|
+
}
|
|
142
|
+
switch (e) {
|
|
143
|
+
case 't': return '\t';
|
|
144
|
+
case 'b': return '\b';
|
|
145
|
+
case 'n': return '\n';
|
|
146
|
+
case 'r': return '\r';
|
|
147
|
+
case 'f': return '\f';
|
|
148
|
+
case '"': return '"';
|
|
149
|
+
case "'": return "'";
|
|
150
|
+
case '\\': return '\\';
|
|
151
|
+
default: return whole;
|
|
152
|
+
}
|
|
153
|
+
});
|
|
154
|
+
// re-emit oxigraph's minimal escaping (escapeNQuadsLiteral): \ " \n \r only.
|
|
155
|
+
return decoded.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\n/g, '\\n').replace(/\r/g, '\\r');
|
|
156
|
+
}
|
|
157
|
+
// Decode N-Triples UCHAR escapes (\uXXXX / \UXXXXXXXX) inside a datatype IRI, as
|
|
158
|
+
// oxigraph does on parse. Out-of-range \U (> U+10FFFF) is left undecoded (oxigraph
|
|
159
|
+
// would reject the literal; we must not throw).
|
|
160
|
+
function decodeIriEscapes(iri) {
|
|
161
|
+
if (!iri.includes('\\'))
|
|
162
|
+
return iri;
|
|
163
|
+
return iri.replace(/\\(u[0-9A-Fa-f]{4}|U[0-9A-Fa-f]{8})/g, (whole, e) => {
|
|
164
|
+
const cp = parseInt(e.slice(1), 16);
|
|
165
|
+
return cp > 0x10ffff ? whole : String.fromCodePoint(cp);
|
|
166
|
+
});
|
|
167
|
+
}
|
|
168
|
+
// oxigraph stores temporal values as seconds-since-0001-01-01 in the same i128/1e18
|
|
169
|
+
// Decimal as xsd:decimal/duration. A date/time whose scaled seconds overflow i128
|
|
170
|
+
// fails to parse and is kept VERBATIM, so a foldable timezone / T24 roll / fraction
|
|
171
|
+
// strip must NOT be applied to it. Replicated here via a proleptic-Gregorian day
|
|
172
|
+
// count. (Byte-exact vs oxigraph for dateTime/date; for the bare g-types the cliff
|
|
173
|
+
// may differ by ≤1 year at the ~5.39e12 boundary — a value impossible in real data.)
|
|
174
|
+
function daysFromCivil(y, m, d) {
|
|
175
|
+
const yy = m <= 2n ? y - 1n : y;
|
|
176
|
+
const era = (yy >= 0n ? yy : yy - 399n) / 400n;
|
|
177
|
+
const yoe = yy - era * 400n;
|
|
178
|
+
const doy = (153n * (m + (m > 2n ? -3n : 9n)) + 2n) / 5n + d - 1n;
|
|
179
|
+
const doe = yoe * 365n + yoe / 4n - yoe / 100n + doy;
|
|
180
|
+
return era * 146097n + doe - 719468n;
|
|
181
|
+
}
|
|
182
|
+
// Inverse of daysFromCivil: proleptic-Gregorian (y,m,d) from a signed day count
|
|
183
|
+
// (days since 1970-01-01). Standard Howard Hinnant algorithm. Used to roll the
|
|
184
|
+
// DATE when a timezone offset pushes a dateTime across midnight during the
|
|
185
|
+
// backend-independent UTC normalization (OT-RFC-57).
|
|
186
|
+
function civilFromDays(zIn) {
|
|
187
|
+
const z = zIn + 719468n;
|
|
188
|
+
const era = (z >= 0n ? z : z - 146096n) / 146097n;
|
|
189
|
+
const doe = z - era * 146097n; // [0, 146096]
|
|
190
|
+
const yoe = (doe - doe / 1460n + doe / 36524n - doe / 146096n) / 365n; // [0, 399]
|
|
191
|
+
const y = yoe + era * 400n;
|
|
192
|
+
const doy = doe - (365n * yoe + yoe / 4n - yoe / 100n); // [0, 365]
|
|
193
|
+
const mp = (5n * doy + 2n) / 153n; // [0, 11]
|
|
194
|
+
const d = doy - (153n * mp + 2n) / 5n + 1n; // [1, 31]
|
|
195
|
+
const m = mp < 10n ? mp + 3n : mp - 9n; // [1, 12]
|
|
196
|
+
return { y: m <= 2n ? y + 1n : y, m, d };
|
|
197
|
+
}
|
|
198
|
+
// OT-RFC-57: the UTC date of "midnight in the given tz" — the backend-independent
|
|
199
|
+
// form for xsd:date / gYear / gYearMonth. Blazegraph interprets the value at 00:00
|
|
200
|
+
// in its tz, converts to UTC, and takes the UTC date; a positive offset rolls the
|
|
201
|
+
// date back a day. offsetMin=0 (Z / no-tz) ⇒ the date is unchanged.
|
|
202
|
+
function utcDateFromMidnight(y, mo, d, offsetMin) {
|
|
203
|
+
const days = daysFromCivil(y, mo, d) + BigInt(Math.floor((0 - offsetMin) / 1440));
|
|
204
|
+
return civilFromDays(days);
|
|
205
|
+
}
|
|
206
|
+
function temporalInRange(yearStr, mo, dd, hh = 0, mi = 0, ss = 0) {
|
|
207
|
+
const seconds = (daysFromCivil(BigInt(yearStr), BigInt(mo), BigInt(dd)) + 719162n) * 86400n +
|
|
208
|
+
BigInt(hh) * 3600n + BigInt(mi) * 60n + BigInt(ss);
|
|
209
|
+
const scaled = seconds * DEC_SCALE;
|
|
210
|
+
return scaled >= I128_MIN && scaled <= I128_MAX;
|
|
211
|
+
}
|
|
212
|
+
// ── xsd:integer family ─────────────────────────────────────────────────────────
|
|
213
|
+
function canonIntegerTerm(lex) {
|
|
214
|
+
if (!/^[+-]?\d+$/.test(lex))
|
|
215
|
+
return null; // at most one sign; "+-1" etc. → verbatim
|
|
216
|
+
const v = BigInt(lex.replace(/^\+/, '')); // BigInt rejects a leading '+'
|
|
217
|
+
if (v < I64_MIN || v > I64_MAX)
|
|
218
|
+
return null; // outside i64 → verbatim (xsd:integer included)
|
|
219
|
+
return `"${v.toString()}"^^<${XSD_INTEGER}>`;
|
|
220
|
+
}
|
|
221
|
+
// ── xsd:boolean ────────────────────────────────────────────────────────────────
|
|
222
|
+
function canonBoolean(lex) {
|
|
223
|
+
if (lex === 'true' || lex === '1')
|
|
224
|
+
return 'true';
|
|
225
|
+
if (lex === 'false' || lex === '0')
|
|
226
|
+
return 'false';
|
|
227
|
+
throw new Error(`invalid xsd:boolean: ${lex}`);
|
|
228
|
+
}
|
|
229
|
+
// ── xsd:decimal ────────────────────────────────────────────────────────────────
|
|
230
|
+
function canonDecimal(lex) {
|
|
231
|
+
const m = /^([+-]?)(\d*)(?:\.(\d*))?$/.exec(lex);
|
|
232
|
+
if (!m || (m[2] === '' && (m[3] === undefined || m[3] === '')))
|
|
233
|
+
throw new Error(`invalid xsd:decimal: ${lex}`);
|
|
234
|
+
const intRaw = m[2].replace(/^0+/, '');
|
|
235
|
+
const frac = (m[3] ?? '').replace(/0+$/, '');
|
|
236
|
+
// oxigraph stores xsd:decimal as the SAME i128 / 10^18 fixed-point as duration
|
|
237
|
+
// seconds: a value needing more than 18 fractional digits, or whose 10^18-scaled
|
|
238
|
+
// magnitude overflows i128, fails to parse and is kept VERBATIM.
|
|
239
|
+
if (frac.length > 18)
|
|
240
|
+
throw new Error('xsd:decimal sub-1e-18');
|
|
241
|
+
const scaled = BigInt((intRaw || '0') + frac.padEnd(18, '0'));
|
|
242
|
+
const signed = m[1] === '-' ? -scaled : scaled;
|
|
243
|
+
if (signed < I128_MIN || signed > I128_MAX)
|
|
244
|
+
throw new Error('xsd:decimal overflow i128');
|
|
245
|
+
const int = intRaw === '' ? '0' : intRaw;
|
|
246
|
+
const sign = m[1] === '-' && !(int === '0' && frac === '') ? '-' : '';
|
|
247
|
+
return frac === '' ? `${sign}${int}` : `${sign}${int}.${frac}`;
|
|
248
|
+
}
|
|
249
|
+
// ── xsd:double / xsd:float ─────────────────────────────────────────────────────
|
|
250
|
+
function canonDouble(lex, isFloat) {
|
|
251
|
+
let n = parseXsdDouble(lex);
|
|
252
|
+
if (isFloat)
|
|
253
|
+
n = Math.fround(n);
|
|
254
|
+
if (Number.isNaN(n))
|
|
255
|
+
return 'NaN';
|
|
256
|
+
if (n === Infinity)
|
|
257
|
+
return 'INF';
|
|
258
|
+
if (n === -Infinity)
|
|
259
|
+
return '-INF';
|
|
260
|
+
// OT-RFC-57: negative zero folds to "0". Blazegraph drops the sign on write
|
|
261
|
+
// ("-0.0"^^double → stored "0.0" → value 0), while oxigraph keeps "-0"; emitting
|
|
262
|
+
// "0" for both signed zeros makes canon(input) == canon(store-readback) on either
|
|
263
|
+
// backend. (The IEEE-754 -0/+0 distinction is not consensus-observable here.)
|
|
264
|
+
if (n === 0)
|
|
265
|
+
return '0';
|
|
266
|
+
const neg = n < 0;
|
|
267
|
+
const a = Math.abs(n);
|
|
268
|
+
// double: V8's a.toString() IS the shortest round-trip; only ties need the
|
|
269
|
+
// away-from-zero correction. float: V8 has no f32-shortest, so search it.
|
|
270
|
+
const shortest = isFloat ? shortestFloat32String(a) : roundTiesAwayFromZero(a, a.toString(), false);
|
|
271
|
+
const plain = expandToPlainDecimal(shortest);
|
|
272
|
+
return neg ? `-${plain}` : plain;
|
|
273
|
+
}
|
|
274
|
+
// V8's Number→string breaks shortest-representation ties round-half-to-EVEN, but
|
|
275
|
+
// oxigraph (Rust) breaks them round-half-AWAY-from-zero. They diverge only when the
|
|
276
|
+
// value sits EXACTLY between two equal-length shortest decimals (e.g. the f64
|
|
277
|
+
// 738507753103385.25 → V8 ".2", Rust ".3"). Detect that tie with exact integer
|
|
278
|
+
// arithmetic and pick the away-from-zero neighbour to match oxigraph.
|
|
279
|
+
function roundTiesAwayFromZero(a, shortest, isFloat) {
|
|
280
|
+
const m = /^(\d+)(?:\.(\d+))?(?:[eE]([+-]?\d+))?$/.exec(shortest);
|
|
281
|
+
if (!m)
|
|
282
|
+
return shortest;
|
|
283
|
+
const digits = m[1] + (m[2] ?? '');
|
|
284
|
+
const D = BigInt(digits);
|
|
285
|
+
const E = (m[3] ? parseInt(m[3], 10) : 0) - (m[2] ? m[2].length : 0); // value = D × 10^E
|
|
286
|
+
const up = D + 1n; // away-from-zero neighbour at the same digit length (a ≥ 0)
|
|
287
|
+
const rt = (x) => (isFloat ? Math.fround(x) : x);
|
|
288
|
+
if (rt(Number(`${up}e${E}`)) !== a)
|
|
289
|
+
return shortest; // up-neighbour doesn't round-trip → no tie
|
|
290
|
+
// Exact tie test: 2·a == (2D+1) × 10^E, with a = num/den from the IEEE-754 bits.
|
|
291
|
+
const [num, den] = f64Fraction(a);
|
|
292
|
+
let lhs = 2n * num;
|
|
293
|
+
let rhs = (2n * D + 1n) * den;
|
|
294
|
+
if (E >= 0)
|
|
295
|
+
rhs *= 10n ** BigInt(E);
|
|
296
|
+
else
|
|
297
|
+
lhs *= 10n ** BigInt(-E);
|
|
298
|
+
return lhs === rhs ? `${up}e${E}` : shortest;
|
|
299
|
+
}
|
|
300
|
+
// Exact value of a finite |f64| as num/den (den a power of two) from its bits.
|
|
301
|
+
function f64Fraction(a) {
|
|
302
|
+
const dv = new DataView(new ArrayBuffer(8));
|
|
303
|
+
dv.setFloat64(0, a);
|
|
304
|
+
const bits = dv.getBigUint64(0);
|
|
305
|
+
const exp = Number((bits >> 52n) & 0x7ffn);
|
|
306
|
+
const fracBits = bits & 0xfffffffffffffn;
|
|
307
|
+
const mant = exp === 0 ? fracBits : fracBits | (1n << 52n);
|
|
308
|
+
const e = (exp === 0 ? -1074 : exp - 1075);
|
|
309
|
+
return e >= 0 ? [mant << BigInt(e), 1n] : [mant, 1n << BigInt(-e)];
|
|
310
|
+
}
|
|
311
|
+
function parseXsdDouble(lex) {
|
|
312
|
+
// oxigraph parses doubles with Rust's lenient f64::from_str: case-INSENSITIVE
|
|
313
|
+
// nan / inf / infinity (with an optional sign) all parse, not just the XSD
|
|
314
|
+
// spellings NaN / INF / -INF. Match it so e.g. "infinity"/"NaN"/"-inf" canon to
|
|
315
|
+
// the oxigraph forms INF / NaN / -INF instead of staying verbatim.
|
|
316
|
+
if (/^[+-]?nan$/i.test(lex))
|
|
317
|
+
return NaN;
|
|
318
|
+
if (/^\+?inf(inity)?$/i.test(lex))
|
|
319
|
+
return Infinity;
|
|
320
|
+
if (/^-inf(inity)?$/i.test(lex))
|
|
321
|
+
return -Infinity;
|
|
322
|
+
if (!/^[+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?$/.test(lex))
|
|
323
|
+
throw new Error(`invalid xsd:double: ${lex}`);
|
|
324
|
+
return Number(lex);
|
|
325
|
+
}
|
|
326
|
+
// Shortest decimal that round-trips to the f32 `a`, matching Rust's f32 formatting.
|
|
327
|
+
// V8 has no native f32-shortest, and a.toPrecision(p)/toExponential round `a` to
|
|
328
|
+
// NEAREST — which can miss the round-tripping p-digit decimal sitting on the other
|
|
329
|
+
// side of `a` (a's f32 rounding interval is wider than its f64 one). So at each
|
|
330
|
+
// precision we test the nearest p-digit mantissa AND its ±1 neighbours (exact
|
|
331
|
+
// integers, no float-grid error), keep those whose f32 round-trip equals a, and
|
|
332
|
+
// pick the closest to a — ties to the away-from-zero (larger) value, as Rust does.
|
|
333
|
+
function shortestFloat32String(a) {
|
|
334
|
+
for (let p = 1; p <= 9; p++) {
|
|
335
|
+
const m = /^(\d)(?:\.(\d+))?e([+-]\d+)$/.exec(a.toExponential(p - 1));
|
|
336
|
+
if (!m)
|
|
337
|
+
break;
|
|
338
|
+
const mant = m[1] + (m[2] ?? ''); // p significant digits
|
|
339
|
+
const e10 = parseInt(m[3], 10) - (p - 1); // value = mant × 10^e10
|
|
340
|
+
const base = BigInt(mant);
|
|
341
|
+
const valid = [];
|
|
342
|
+
for (const v of [base, base - 1n, base + 1n]) {
|
|
343
|
+
if (v <= 0n)
|
|
344
|
+
continue;
|
|
345
|
+
const c = Number(`${v}e${e10}`);
|
|
346
|
+
if (Math.fround(c) === a)
|
|
347
|
+
valid.push(c);
|
|
348
|
+
}
|
|
349
|
+
if (valid.length) {
|
|
350
|
+
valid.sort((x, y) => Math.abs(x - a) - Math.abs(y - a) || y - x); // closest; tie → away from zero
|
|
351
|
+
return valid[0].toString();
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
return a.toString();
|
|
355
|
+
}
|
|
356
|
+
function expandToPlainDecimal(s) {
|
|
357
|
+
const m = /^(\d+)(?:\.(\d+))?[eE]([+-]?\d+)$/.exec(s);
|
|
358
|
+
if (!m)
|
|
359
|
+
return s;
|
|
360
|
+
const intPart = m[1];
|
|
361
|
+
const frac = m[2] ?? '';
|
|
362
|
+
const exp = parseInt(m[3], 10);
|
|
363
|
+
const digits = intPart + frac;
|
|
364
|
+
const pointPos = intPart.length + exp;
|
|
365
|
+
if (pointPos <= 0)
|
|
366
|
+
return stripTrailingZeros(`0.${'0'.repeat(-pointPos)}${digits}`);
|
|
367
|
+
if (pointPos >= digits.length)
|
|
368
|
+
return digits + '0'.repeat(pointPos - digits.length);
|
|
369
|
+
return stripTrailingZeros(`${digits.slice(0, pointPos)}.${digits.slice(pointPos)}`);
|
|
370
|
+
}
|
|
371
|
+
function stripTrailingZeros(s) {
|
|
372
|
+
if (!s.includes('.'))
|
|
373
|
+
return s;
|
|
374
|
+
return s.replace(/0+$/, '').replace(/\.$/, '');
|
|
375
|
+
}
|
|
376
|
+
// ── date/time family ───────────────────────────────────────────────────────────
|
|
377
|
+
// Returns the offset MAGNITUDE in minutes (signed) for the
|
|
378
|
+
// backend-independent UTC normalization of xsd:dateTime/xsd:time (OT-RFC-57).
|
|
379
|
+
// hadTz=false ⇒ no timezone present (a bare dateTime is normalized to UTC and
|
|
380
|
+
// gains a Z, matching Blazegraph/Neptune). Malformed/out-of-range tz → throw
|
|
381
|
+
// (→ the literal is kept verbatim, as oxigraph does).
|
|
382
|
+
function splitTzToOffset(s) {
|
|
383
|
+
const m = /(Z|[+-]\d{2}:\d{2})$/.exec(s);
|
|
384
|
+
if (!m)
|
|
385
|
+
return { body: s, offsetMin: 0, hadTz: false };
|
|
386
|
+
const tz = m[1];
|
|
387
|
+
const body = s.slice(0, s.length - tz.length);
|
|
388
|
+
if (tz === 'Z')
|
|
389
|
+
return { body, offsetMin: 0, hadTz: true };
|
|
390
|
+
const h = parseInt(tz.slice(1, 3), 10);
|
|
391
|
+
const mi = parseInt(tz.slice(4, 6), 10);
|
|
392
|
+
if (mi > 59 || h * 60 + mi > 840)
|
|
393
|
+
throw new Error(`invalid tz: ${tz}`);
|
|
394
|
+
const mag = h * 60 + mi;
|
|
395
|
+
return { body, offsetMin: tz[0] === '-' ? -mag : mag, hadTz: true };
|
|
396
|
+
}
|
|
397
|
+
// Normalize a fractional-seconds group ('.ddd' or undefined): TRUNCATE to at most
|
|
398
|
+
// 3 digits (milliseconds — the backend-independent precision floor; a lossy store
|
|
399
|
+
// such as Blazegraph keeps only ms), then strip trailing zeros; drop entirely if
|
|
400
|
+
// empty. Truncate, NOT round (matches Blazegraph). (OT-RFC-57)
|
|
401
|
+
function normFrac(frac) {
|
|
402
|
+
if (frac === undefined)
|
|
403
|
+
return '';
|
|
404
|
+
const d = frac.slice(1, 4).replace(/0+$/, ''); // at most 3 digits, then strip trailing zeros
|
|
405
|
+
return d === '' ? '' : `.${d}`;
|
|
406
|
+
}
|
|
407
|
+
// Proleptic-Gregorian leap test on the astronomical year number (handles negative
|
|
408
|
+
// and arbitrarily large years via BigInt).
|
|
409
|
+
function isLeapYear(yearStr) {
|
|
410
|
+
const y = BigInt(yearStr);
|
|
411
|
+
return (y % 4n === 0n && y % 100n !== 0n) || y % 400n === 0n;
|
|
412
|
+
}
|
|
413
|
+
function daysInMonth(yearStr, mo) {
|
|
414
|
+
return [31, isLeapYear(yearStr) ? 29 : 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31][mo - 1];
|
|
415
|
+
}
|
|
416
|
+
// Day after (yearStr, mo, dd). Year crosses are computed on the signed numeric
|
|
417
|
+
// year (BigInt) then re-emitted min-4-digit, sign-preserved (…→0000, 9999→10000).
|
|
418
|
+
function rollNextDay(yearStr, mo, dd) {
|
|
419
|
+
let ny = BigInt(yearStr);
|
|
420
|
+
let nmo = mo;
|
|
421
|
+
let nd = dd + 1;
|
|
422
|
+
if (nd > daysInMonth(yearStr, mo)) {
|
|
423
|
+
nd = 1;
|
|
424
|
+
nmo += 1;
|
|
425
|
+
if (nmo > 12) {
|
|
426
|
+
nmo = 1;
|
|
427
|
+
ny += 1n;
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
return `${fmtYear(ny)}-${pad2(nmo)}-${pad2(nd)}`;
|
|
431
|
+
}
|
|
432
|
+
function fmtYear(y) {
|
|
433
|
+
const neg = y < 0n;
|
|
434
|
+
const abs = (neg ? -y : y).toString().padStart(4, '0');
|
|
435
|
+
return neg ? `-${abs}` : abs;
|
|
436
|
+
}
|
|
437
|
+
const pad2 = (n) => String(n).padStart(2, '0');
|
|
438
|
+
// hh∈[0,24], mm∈[0,59], ss∈[0,59]; hour 24 is valid ONLY when it can roll, which
|
|
439
|
+
// oxigraph allows iff minute==0 OR the seconds value (incl. fraction) is 0.
|
|
440
|
+
// Returns whether the time rolls to the next day (hour 24 → 00). Throws on any
|
|
441
|
+
// out-of-range field or a non-rollable hour-24.
|
|
442
|
+
function validateClock(hh, mi, ss, fracNorm) {
|
|
443
|
+
if (hh > 24 || mi > 59 || ss > 59)
|
|
444
|
+
throw new Error('clock out of range');
|
|
445
|
+
if (hh === 24) {
|
|
446
|
+
const secZero = ss === 0 && fracNorm === '';
|
|
447
|
+
if (mi !== 0 && !secZero)
|
|
448
|
+
throw new Error('invalid hour-24');
|
|
449
|
+
return { rolls: true };
|
|
450
|
+
}
|
|
451
|
+
return { rolls: false };
|
|
452
|
+
}
|
|
453
|
+
// OT-RFC-57: the backend-independent value canon accepts any 4+-digit year (any
|
|
454
|
+
// number of leading zeros) and normalizes it via BigInt+fmtYear (min-4-digit, no
|
|
455
|
+
// leading zero). This matches Blazegraph, which on write STRIPS a leading-zero
|
|
456
|
+
// year to its value ("02026"^^gYear → "2026") — oxigraph instead keeps the invalid
|
|
457
|
+
// literal verbatim, but the CONVERGENCE oracle holds either way since canon(input)
|
|
458
|
+
// and canon(store-readback) both fold to the same value form (OT-RFC-57 §7.5).
|
|
459
|
+
const YEAR = '-?\\d{4,}';
|
|
460
|
+
// OT-RFC-57 backend-independent form: normalize to UTC (subtract the tz offset,
|
|
461
|
+
// rolling the DATE across midnight), truncate fraction to ms, always emit Z. A
|
|
462
|
+
// no-timezone dateTime is treated as UTC and gains a Z (matching Blazegraph /
|
|
463
|
+
// Neptune). This is the value-space form the publisher's input AND every
|
|
464
|
+
// backend's read-back converge to.
|
|
465
|
+
function canonDateTime(lex) {
|
|
466
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
467
|
+
const m = new RegExp(`^(${YEAR})-(\\d{2})-(\\d{2})T(\\d{2}):(\\d{2}):(\\d{2})(\\.\\d+)?$`).exec(body);
|
|
468
|
+
if (!m)
|
|
469
|
+
throw new Error('invalid xsd:dateTime');
|
|
470
|
+
const [, yy, mo, dd, hh, mi, ss, frac] = m;
|
|
471
|
+
const moN = +mo;
|
|
472
|
+
const ddN = +dd;
|
|
473
|
+
if (moN < 1 || moN > 12)
|
|
474
|
+
throw new Error('month');
|
|
475
|
+
if (ddN < 1 || ddN > daysInMonth(yy, moN))
|
|
476
|
+
throw new Error('day');
|
|
477
|
+
const fracNorm = normFrac(frac);
|
|
478
|
+
const { rolls } = validateClock(+hh, +mi, +ss, fracNorm);
|
|
479
|
+
// Base date as a day count; a T24:00 clock rolls one day and resets the hour to 0.
|
|
480
|
+
let days = daysFromCivil(BigInt(yy), BigInt(moN), BigInt(ddN));
|
|
481
|
+
const hourN = rolls ? 0 : +hh;
|
|
482
|
+
if (rolls)
|
|
483
|
+
days += 1n;
|
|
484
|
+
// UTC: subtract the offset (whole minutes); roll the date across midnight.
|
|
485
|
+
const totalMin = hourN * 60 + +mi - offsetMin;
|
|
486
|
+
days += BigInt(Math.floor(totalMin / 1440));
|
|
487
|
+
const minInDay = ((totalMin % 1440) + 1440) % 1440;
|
|
488
|
+
const { y, m: mm, d } = civilFromDays(days);
|
|
489
|
+
// Range-check the NORMALIZED UTC instant, not the lexical components: a tz offset
|
|
490
|
+
// or T24 roll can push a boundary value outside the i128 seconds range it would
|
|
491
|
+
// otherwise pass, emitting a leaf for a value the store can't represent stably
|
|
492
|
+
// (otReviewAgent). Out of range → verbatim (throw, caught upstream).
|
|
493
|
+
if (!temporalInRange(y.toString(), Number(mm), Number(d), Math.floor(minInDay / 60), minInDay % 60, +ss))
|
|
494
|
+
throw new Error('normalized dateTime overflows i128 seconds');
|
|
495
|
+
return `${fmtYear(y)}-${pad2(Number(mm))}-${pad2(Number(d))}T${pad2(Math.floor(minInDay / 60))}:${pad2(minInDay % 60)}:${ss}${fracNorm}Z`;
|
|
496
|
+
}
|
|
497
|
+
// OT-RFC-57: time has no date, so a tz offset just wraps the wall clock mod 24h;
|
|
498
|
+
// normalize to UTC + Z, ms-truncated.
|
|
499
|
+
function canonTime(lex) {
|
|
500
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
501
|
+
const m = /^(\d{2}):(\d{2}):(\d{2})(\.\d+)?$/.exec(body);
|
|
502
|
+
if (!m)
|
|
503
|
+
throw new Error('invalid xsd:time');
|
|
504
|
+
const [, hh, mi, ss, frac] = m;
|
|
505
|
+
const fracNorm = normFrac(frac);
|
|
506
|
+
const { rolls } = validateClock(+hh, +mi, +ss, fracNorm);
|
|
507
|
+
const hourN = rolls ? 0 : +hh;
|
|
508
|
+
const minInDay = (((hourN * 60 + +mi - offsetMin) % 1440) + 1440) % 1440;
|
|
509
|
+
return `${pad2(Math.floor(minInDay / 60))}:${pad2(minInDay % 60)}:${ss}${fracNorm}Z`;
|
|
510
|
+
}
|
|
511
|
+
// OT-RFC-57: xsd:date / gYear / gYearMonth normalize to the UTC date of
|
|
512
|
+
// midnight-in-tz, with NO timezone emitted (Blazegraph's value form).
|
|
513
|
+
function canonDate(lex) {
|
|
514
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
515
|
+
const m = new RegExp(`^(${YEAR})-(\\d{2})-(\\d{2})$`).exec(body);
|
|
516
|
+
if (!m)
|
|
517
|
+
throw new Error('invalid xsd:date');
|
|
518
|
+
const moN = +m[2];
|
|
519
|
+
const ddN = +m[3];
|
|
520
|
+
if (moN < 1 || moN > 12)
|
|
521
|
+
throw new Error('month');
|
|
522
|
+
if (ddN < 1 || ddN > daysInMonth(m[1], moN))
|
|
523
|
+
throw new Error('day');
|
|
524
|
+
const { y, m: mm, d } = utcDateFromMidnight(BigInt(m[1]), BigInt(moN), BigInt(ddN), offsetMin);
|
|
525
|
+
// Validate the NORMALIZED date (the tz roll can cross the year boundary) — see canonDateTime.
|
|
526
|
+
if (!temporalInRange(y.toString(), Number(mm), Number(d)))
|
|
527
|
+
throw new Error('normalized date overflows i128 seconds');
|
|
528
|
+
return `${fmtYear(y)}-${pad2(Number(mm))}-${pad2(Number(d))}`;
|
|
529
|
+
}
|
|
530
|
+
function canonGYear(lex) {
|
|
531
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
532
|
+
if (!new RegExp(`^${YEAR}$`).test(body))
|
|
533
|
+
throw new Error('invalid xsd:gYear');
|
|
534
|
+
const { y, m: mm, d } = utcDateFromMidnight(BigInt(body), 1n, 1n, offsetMin);
|
|
535
|
+
// Validate the NORMALIZED date (a negative offset can roll 01-01 into the prior year).
|
|
536
|
+
if (!temporalInRange(y.toString(), Number(mm), Number(d)))
|
|
537
|
+
throw new Error('normalized gYear overflows i128 seconds');
|
|
538
|
+
return fmtYear(y);
|
|
539
|
+
}
|
|
540
|
+
function canonGYearMonth(lex) {
|
|
541
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
542
|
+
const m = new RegExp(`^(${YEAR})-(\\d{2})$`).exec(body);
|
|
543
|
+
if (!m || +m[2] < 1 || +m[2] > 12)
|
|
544
|
+
throw new Error('invalid xsd:gYearMonth');
|
|
545
|
+
const { y, m: mm, d } = utcDateFromMidnight(BigInt(m[1]), BigInt(+m[2]), 1n, offsetMin);
|
|
546
|
+
// Validate the NORMALIZED date (the tz roll can cross the year boundary).
|
|
547
|
+
if (!temporalInRange(y.toString(), Number(mm), Number(d)))
|
|
548
|
+
throw new Error('normalized gYearMonth overflows i128 seconds');
|
|
549
|
+
return `${fmtYear(y)}-${pad2(Number(mm))}`;
|
|
550
|
+
}
|
|
551
|
+
// gMonthDay day bounds. oxigraph 0.5.5 validates --MM-DD against a NON-leap
|
|
552
|
+
// reference year, so --02-29 is rejected (kept verbatim) — February's max is 28
|
|
553
|
+
// here, unlike a real leap date which needs the year context of xsd:date.
|
|
554
|
+
const MONTH_MAX_DAY = [31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
|
|
555
|
+
// OT-RFC-57: gMonthDay / gMonth / gDay have no year/date context to convert a
|
|
556
|
+
// timezone into UTC. We therefore fold ONLY a UTC-equivalent zone (Z / +00:00 /
|
|
557
|
+
// -00:00 → offsetMin 0) to the no-timezone value form. A NON-UTC offset is kept
|
|
558
|
+
// VERBATIM (the whole literal, offset included): stripping it would silently
|
|
559
|
+
// COLLAPSE distinct values — "--06-29+14:00" and "--06-29-14:00" are different
|
|
560
|
+
// literals — onto one leaf (otReviewAgent). Verbatim keeps them distinct and defers
|
|
561
|
+
// to the store's own preservation; such exotic offsets on bare gregorian types are
|
|
562
|
+
// vanishingly rare and out of the consensus-verified set (see OT-RFC-57 §7.8).
|
|
563
|
+
function bareGregorian(lex, re, validate) {
|
|
564
|
+
const { body, offsetMin } = splitTzToOffset(lex);
|
|
565
|
+
const m = re.exec(body);
|
|
566
|
+
if (!m || !validate(m))
|
|
567
|
+
throw new Error('invalid bare gregorian');
|
|
568
|
+
return offsetMin === 0 ? body : lex; // fold UTC-equivalent zone only; else verbatim
|
|
569
|
+
}
|
|
570
|
+
function canonGMonthDay(lex) {
|
|
571
|
+
return bareGregorian(lex, /^--(\d{2})-(\d{2})$/, (m) => {
|
|
572
|
+
const moN = +m[1];
|
|
573
|
+
const ddN = +m[2];
|
|
574
|
+
return moN >= 1 && moN <= 12 && ddN >= 1 && ddN <= MONTH_MAX_DAY[moN - 1];
|
|
575
|
+
});
|
|
576
|
+
}
|
|
577
|
+
function canonGMonth(lex) {
|
|
578
|
+
return bareGregorian(lex, /^--(\d{2})$/, (m) => +m[1] >= 1 && +m[1] <= 12);
|
|
579
|
+
}
|
|
580
|
+
function canonGDay(lex) {
|
|
581
|
+
return bareGregorian(lex, /^---(\d{2})$/, (m) => +m[1] >= 1 && +m[1] <= 31);
|
|
582
|
+
}
|
|
583
|
+
// ── xsd:duration / dayTimeDuration / yearMonthDuration ─────────────────────────
|
|
584
|
+
// Value-space canonicalization mirroring oxsdatatypes Duration { months: i64,
|
|
585
|
+
// seconds: Decimal(i128 / 10^18) }. Parse to (months, scaledSeconds), reject if
|
|
586
|
+
// either overflows its integer type (→ verbatim), then re-emit the canonical
|
|
587
|
+
// component breakdown (Y=months/12, M=months%12; D/H/M/S from seconds).
|
|
588
|
+
// Seconds accept a trailing dot with no fraction (oxigraph: "PT1.S" → "PT1S") and a
|
|
589
|
+
// leading dot ("PT.5S" → "PT0.5S"), matching Rust's lenient parse.
|
|
590
|
+
const RE_DURATION = /^(-?)P(?:(\d+)Y)?(?:(\d+)M)?(?:(\d+)D)?(?:T(?:(\d+)H)?(?:(\d+)M)?(?:(\d+(?:\.\d*)?|\.\d+)S)?)?$/;
|
|
591
|
+
function canonDuration(lex, dt) {
|
|
592
|
+
const m = RE_DURATION.exec(lex);
|
|
593
|
+
if (!m)
|
|
594
|
+
throw new Error(`invalid duration: ${lex}`);
|
|
595
|
+
const [, sign, y, mo, d, h, mi, sTok] = m;
|
|
596
|
+
if (!y && !mo && !d && !h && !mi && !sTok)
|
|
597
|
+
throw new Error('empty duration'); // "P" / "PT"
|
|
598
|
+
const isYM = dt === XSD + 'yearMonthDuration';
|
|
599
|
+
const isDT = dt === XSD + 'dayTimeDuration';
|
|
600
|
+
// Subtype component constraints (oxigraph keeps a mis-componented subtype verbatim).
|
|
601
|
+
if (isYM && (d || h || mi || sTok))
|
|
602
|
+
throw new Error('yearMonthDuration: time/day component');
|
|
603
|
+
if (isDT && (y || mo))
|
|
604
|
+
throw new Error('dayTimeDuration: year/month component');
|
|
605
|
+
// Seconds token → whole + fractional (≤18 significant fractional digits; the
|
|
606
|
+
// oxsdatatypes Decimal scale is 10^18, so a 19th significant digit is rejected).
|
|
607
|
+
let sWhole = '0';
|
|
608
|
+
let fracScaled = 0n;
|
|
609
|
+
if (sTok) {
|
|
610
|
+
const dot = sTok.indexOf('.');
|
|
611
|
+
if (dot === -1) {
|
|
612
|
+
sWhole = sTok;
|
|
613
|
+
}
|
|
614
|
+
else {
|
|
615
|
+
sWhole = sTok.slice(0, dot) || '0';
|
|
616
|
+
const fracDigits = sTok.slice(dot + 1).replace(/0+$/, '');
|
|
617
|
+
if (fracDigits.length > 18)
|
|
618
|
+
throw new Error('sub-1e-18 seconds');
|
|
619
|
+
fracScaled = fracDigits === '' ? 0n : BigInt(fracDigits.padEnd(18, '0'));
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
const months = 12n * BigInt(y || '0') + BigInt(mo || '0');
|
|
623
|
+
const wholeSeconds = 86400n * BigInt(d || '0') + 3600n * BigInt(h || '0') + 60n * BigInt(mi || '0') + BigInt(sWhole);
|
|
624
|
+
const scaledSeconds = wholeSeconds * DEC_SCALE + fracScaled;
|
|
625
|
+
// Overflow at oxigraph's stored-integer boundaries (signed) → verbatim.
|
|
626
|
+
const neg = sign === '-';
|
|
627
|
+
const signedMonths = neg ? -months : months;
|
|
628
|
+
const signedScaled = neg ? -scaledSeconds : scaledSeconds;
|
|
629
|
+
if (signedMonths < I64_MIN || signedMonths > I64_MAX)
|
|
630
|
+
throw new Error('months overflow i64');
|
|
631
|
+
if (signedScaled < I128_MIN || signedScaled > I128_MAX)
|
|
632
|
+
throw new Error('seconds overflow i128');
|
|
633
|
+
// Re-derive canonical components from magnitudes (sign emitted once).
|
|
634
|
+
const yy = months / 12n;
|
|
635
|
+
const MM = months % 12n;
|
|
636
|
+
const totalWhole = scaledSeconds / DEC_SCALE;
|
|
637
|
+
const fracRem = scaledSeconds % DEC_SCALE;
|
|
638
|
+
const D = totalWhole / 86400n;
|
|
639
|
+
let rem = totalWhole % 86400n;
|
|
640
|
+
const H = rem / 3600n;
|
|
641
|
+
rem %= 3600n;
|
|
642
|
+
const Min = rem / 60n;
|
|
643
|
+
const S = rem % 60n;
|
|
644
|
+
let date = '';
|
|
645
|
+
if (yy > 0n)
|
|
646
|
+
date += `${yy}Y`;
|
|
647
|
+
if (MM > 0n)
|
|
648
|
+
date += `${MM}M`;
|
|
649
|
+
if (D > 0n)
|
|
650
|
+
date += `${D}D`;
|
|
651
|
+
let time = '';
|
|
652
|
+
if (H > 0n)
|
|
653
|
+
time += `${H}H`;
|
|
654
|
+
if (Min > 0n)
|
|
655
|
+
time += `${Min}M`;
|
|
656
|
+
if (S > 0n || fracRem > 0n) {
|
|
657
|
+
const fracStr = fracRem === 0n ? '' : `.${fracRem.toString().padStart(18, '0').replace(/0+$/, '')}`;
|
|
658
|
+
time += `${S}${fracStr}S`;
|
|
659
|
+
}
|
|
660
|
+
const body = time ? `${date}T${time}` : date;
|
|
661
|
+
// All-zero canonical form is subtype-dependent: yearMonthDuration → "P0M",
|
|
662
|
+
// duration / dayTimeDuration → "PT0S".
|
|
663
|
+
if (body === '')
|
|
664
|
+
return isYM ? 'P0M' : 'PT0S';
|
|
665
|
+
return `${neg ? '-' : ''}P${body}`;
|
|
666
|
+
}
|
|
667
|
+
//# sourceMappingURL=term-canon.js.map
|