@gmb/bitmark-parser 6.11.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +222 -54
  2. package/dist/browser/bitmark-parser.min.js +4 -4
  3. package/dist/browser/bitmark-parser.min.js.map +1 -1
  4. package/dist/browser/cjs/index.cjs +547 -325
  5. package/dist/browser/cjs/index.cjs.map +1 -1
  6. package/dist/browser/cjs/index.d.cts +342 -68
  7. package/dist/browser/esm/index.d.ts +342 -68
  8. package/dist/browser/esm/index.js +546 -324
  9. package/dist/browser/esm/index.js.map +1 -1
  10. package/dist/browser/esm/worker-entry.js +497 -242
  11. package/dist/browser/esm/worker-entry.js.map +1 -1
  12. package/dist/browser/wasm/bitmark_browser_full_wasm_bg.wasm +0 -0
  13. package/dist/browser/wasm/bitmark_json_wasm_bg.wasm +0 -0
  14. package/dist/browser/wasm/bitmark_wasm_bg.wasm +0 -0
  15. package/dist/index.cjs +51 -88
  16. package/dist/index.cjs.map +1 -1
  17. package/dist/index.d.cts +342 -68
  18. package/dist/index.d.ts +342 -68
  19. package/dist/index.js +50 -87
  20. package/dist/index.js.map +1 -1
  21. package/dist/legacy.cjs +8 -30
  22. package/dist/legacy.cjs.map +1 -1
  23. package/dist/legacy.d.cts +0 -2
  24. package/dist/legacy.d.ts +0 -2
  25. package/dist/legacy.js +8 -30
  26. package/dist/legacy.js.map +1 -1
  27. package/dist/worker-entry.cjs +1 -5
  28. package/dist/worker-entry.cjs.map +1 -1
  29. package/package.json +6 -6
  30. package/schema/bitmark.schema.json +496 -139
  31. package/wasm/bitmark_wasm.d.ts +27 -23
  32. package/wasm/bitmark_wasm.js +198 -112
  33. package/wasm/bitmark_wasm_bg.wasm +0 -0
  34. package/wasm/bitmark_wasm_bg.wasm.d.ts +2 -2
  35. package/wasm/package.json +1 -1
  36. package/wasm-bitmark-json/bitmark_json_wasm.d.ts +18 -12
  37. package/wasm-bitmark-json/bitmark_json_wasm.js +170 -72
  38. package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm +0 -0
  39. package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm.d.ts +1 -1
  40. package/wasm-bitmark-json/package.json +1 -1
  41. package/wasm-browser-full/bitmark_browser_full_wasm.d.ts +27 -23
  42. package/wasm-browser-full/bitmark_browser_full_wasm.js +198 -112
  43. package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm +0 -0
  44. package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm.d.ts +2 -2
  45. package/wasm-browser-full/package.json +1 -1
package/README.md CHANGED
@@ -11,7 +11,7 @@ npm install @gmb/bitmark-parser
11
11
  ```js
12
12
  import {
13
13
  convert,
14
- lex,
14
+ semanticTokens,
15
15
  breakscapeText,
16
16
  unbreakscapeText,
17
17
  info,
@@ -32,11 +32,18 @@ console.log(parsedJson[0].bitmark);
32
32
  // Generate bitmark from JSON
33
33
  const bitmark = convert(json, { inputFormat: "json", outputFormat: "bitmark" });
34
34
 
35
- // Lex bitmark to token dump
36
- const tokens = lex("[.article]\nHello **bold**");
35
+ // Parser-derived highlighting in the LSP semantic-tokens shape
36
+ const highlights = semanticTokens("[.article]\nHello **bold**", {
37
+ tokensLayout: "absolute",
38
+ });
37
39
 
38
- // Lex with a specific stage
39
- const tokenJson = lex("[.article]\nHello **bold**", { stage: "lex-json" });
40
+ // Dump the lexer's token stream (debug tooling; bitmark input only)
41
+ const tokens = JSON.parse(
42
+ convert("[.article]\nHello **bold**", {
43
+ inputFormat: "bitmark",
44
+ outputFormat: "lex",
45
+ }),
46
+ );
40
47
 
41
48
  // Breakscape / unbreakscape text
42
49
  const escaped = breakscapeText("[!example]", {
@@ -60,23 +67,37 @@ console.log(version());
60
67
  The package ships **three wasm builds of the same engine** and selects one at
61
68
  runtime — the API surface never changes:
62
69
 
63
- | Feature | Contents | Size (approx.) |
64
- | --- | --- | --- |
65
- | `full` (Node default) | everything: rich `info` metadata + built-in translations + the semantic `diff` | ~1082 KB (~423 KB gzip) |
66
- | `browser-full` (browser default) | the same conversions and `diff`, without either | ~935 KB (~374 KB gzip) |
67
- | `bitmark-json` | bitmark ↔ JSON only: `convert`/`canonicalize`/`transform` (formats `auto`/`bitmark`/`json`, plus the `text` output), `info`, breakscape, text fragments — no `diff` | ~705 KB (~277 KB gzip) |
68
-
69
- Not in `bitmark-json`: the markup formats (`html`, `xml`, `xml-niso-iec`, …)
70
- as input or output (they return an `error: … not supported in this build`
71
- string), the `mappingReport` option (same error), and `lex` (throws
72
- `UnsupportedFeatureError`).
70
+ | Feature | Contents | Size (approx.) |
71
+ | -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------- |
72
+ | `full` (Node default) | everything: rich `info` metadata + built-in translations + the semantic `diff` | ~990 KB (~378 KB gzip) |
73
+ | `browser-full` (browser default) | the same conversions and `diff`, without either | ~783 KB (~317 KB gzip) |
74
+ | `bitmark-json` | bitmark ↔ JSON only: `convert`/`canonicalize`/`transform` (formats `auto`/`bitmark`/`json`, plus the `text` and `semantic-tokens` outputs), `info` (JSON only), breakscape, text fragments — no `diff` | ~563 KB (~224 KB gzip) |
75
+
76
+ **The rule.** _Every build includes every feature, except the two
77
+ browser-targeted variants, which each carry an explicit exclusion list._ That
78
+ is the whole story — a new capability is in every build unless it is named
79
+ below:
80
+
81
+ | build | features |
82
+ | -------------------------------------------------- | ----------------------------------------------------------------------------------- |
83
+ | native CLI, daemon, docgen, schemagen, wasm `full` | **all** |
84
+ | wasm `browser-full` | all **except** `info-meta` (bit + tag prose) and `translations` (the baked table) |
85
+ | wasm `bitmark-json` | all **except** the above and `markup`, `mapping-report`, `diff`, `lex`, `info-text` |
86
+
87
+ So, not in `bitmark-json`: the markup formats (`html`, `xml`,
88
+ `xml-niso-iec`, …) as input or output (they throw an `unsupported` error), the
89
+ `lex` output format and the `mappingReport` option (same error), `diff`
90
+ (throws `UnsupportedFeatureError`), and the human-readable rendering of `info` —
91
+ `info({ format: "text" })` throws `UnsupportedFeatureError` there, so pass
92
+ `format: "json"` (the JSON views are the contractual ones and are in every
93
+ variant).
73
94
 
74
95
  `full` and `browser-full` convert **identically** — they differ only in what
75
- `info` can report: the META fields (tag descriptions, group provenance and raw
76
- mapping patterns) and the built-in translations table. Bit titles, bit
77
- descriptions and the bit-group / resource-group catalogs are in every variant,
78
- in English; `register` supplies translations to the lean ones (see *Display
79
- names and languages*).
96
+ `info` can report: the META fields (bit and tag descriptions, group provenance
97
+ and raw mapping patterns) and the built-in translations table. Bit titles and
98
+ the bit-group / resource-group catalogs are in every variant, in English;
99
+ `register` supplies translations to the lean ones (see _Display names and
100
+ languages_).
80
101
 
81
102
  Why the defaults differ: only the browser pays a download-latency cost for
82
103
  those strings, and a Node backend loading wasm from disk has the same
@@ -89,15 +110,15 @@ browser, select `browser-full` explicitly.**
89
110
  > **Migrating from ≤ 6.10:** `full` used to mean what `browser-full` means now.
90
111
  > Browser code that explicitly selected `"full"` should move to
91
112
  > `"browser-full"` to keep its download size close to what it was; leave it on
92
- > `"full"` to gain the metadata and built-in translations for ~49 KB gzip.
113
+ > `"full"` to gain the metadata and built-in translations for ~62 KB gzip.
93
114
  > Nothing else changes — the capabilities that were in `full` are in both.
94
115
 
95
116
  ```js
96
117
  import { init, convert, variant } from "@gmb/bitmark-parser";
97
118
 
98
119
  await init({ feature: "bitmark-json" }); // load the minimal wasm
99
- convert("[.article]\nHello"); // works
100
- variant(); // "bitmark-json"
120
+ convert("[.article]\nHello"); // works
121
+ variant(); // "bitmark-json"
101
122
 
102
123
  // Upgrade in place: the interface stays valid while the full wasm loads —
103
124
  // calls keep being served by the active module, then swap atomically.
@@ -130,6 +151,39 @@ Notes:
130
151
 
131
152
  ### API
132
153
 
154
+ #### Errors
155
+
156
+ **Errors are thrown, never returned.** Every fallible WASM export raises a
157
+ JavaScript `Error` whose `message` has one shape:
158
+
159
+ ```text
160
+ <kind> at <path>: <message>
161
+ ```
162
+
163
+ A returned string is therefore always a result — there is no prefix or
164
+ sentinel to test for, and every call site is a `try`/`catch`.
165
+
166
+ `<kind>` is a **stable kebab-case identifier** — the part to match on.
167
+ `<path>` locates the failure (a JSON pointer-ish path, `offset N` for a JSON
168
+ syntax error, or empty) and `<message>` is human-readable prose that may
169
+ change in any release. The kinds:
170
+
171
+ | kind | meaning |
172
+ | -------------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
173
+ | `invalid-json` | the input string is not well-formed JSON |
174
+ | `compose` | the reverse composer rejected the JSON (shape / field / type) |
175
+ | `unsupported` | the requested conversion is not available (an output-only format used as input, or a format compiled out of this variant) |
176
+ | `patch-parse-error` | a patch or patch document is malformed |
177
+ | `patch-index-out-of-range` | a patch path addressed an array element that does not exist |
178
+ | `patch-type-mismatch` | a patch value's JSON type is incompatible with its target |
179
+ | `patch-conflict` | a patch document's entries contradict each other (a bit removed or moved twice, or an anchor that is itself removed or moved) |
180
+
181
+ Every TypeScript wrapper propagates that `Error` unchanged — `convert`,
182
+ `transform`, `canonicalize`, `bitmarkToObjects`, `countBits`, `diff`, … — as
183
+ does the parallel worker pool, which rejects with the first failing bit's
184
+ error. `register` re-wraps it as a `RegisterError`; a capability the loaded
185
+ variant compiled out raises `UnsupportedFeatureError` before any wasm call.
186
+
133
187
  #### convert(input: string, options?: ConvertOptions): string
134
188
 
135
189
  Convert between bitmark, JSON, and the mapped markup formats (html/xml/…).
@@ -153,7 +207,7 @@ Options:
153
207
  the bit is `_`-prefixed. A bare tag the bit does not declare (`[!…]`,
154
208
  `[?…]`, `[#…]`, … on a bit whose config has no such tag) is an unknown
155
209
  too, under the key a declaring bit would use (`[!x]` → `"instruction":
156
- ["x"]`); a rejected resource attachment is not — it is still emitted, with
210
+ ["x"]`); a rejected resource attachment is not — it is still emitted, with
157
211
  a warning. Unknown properties are never converted back to bitmark, and
158
212
  each occurrence warns whether or not it is included.
159
213
  Also available on `bitmarkToObjects` and as the CLI's
@@ -175,6 +229,13 @@ Options:
175
229
  > format, set `inputFormat` explicitly. The same applies to `canonicalize`
176
230
  > and `transform`, whose `inputFormat` also defaults to `"auto"`.
177
231
 
232
+ > **Input limits:** JSON objects/arrays and markup elements may nest at most
233
+ > **256** levels deep. Deeper JSON is rejected like any other malformed JSON
234
+ > (`"error: … nesting depth exceeds 256"`); a deeper markup open tag is kept
235
+ > as literal text. The bound keeps a crafted document from exhausting the
236
+ > WASM stack — an instance that rejects a too-deep input stays usable for the
237
+ > next call.
238
+
178
239
  `convert` is the single string-based conversion surface — parsing is
179
240
  `{ inputFormat: "bitmark", outputFormat: "json" }`, generating is
180
241
  `{ inputFormat: "json", outputFormat: "bitmark" }`. For typed objects use
@@ -184,7 +245,15 @@ For bitmark input, output is a JSON array of entries with:
184
245
 
185
246
  - `bit` — serialized bit payload
186
247
  - `parser` — parser metadata, `errors` / `warnings`, and `infos` (informational notices for recoveries that are not necessarily wrong, e.g. text kept as text because it only looks like an unclosed tag). Each issue carries a stable machine-readable `code` (`"unknown-property"`, `"missing-required-tag"`, …) beside its human-readable `message`, plus `text` and `location`. **Branch on `code`** — a shipped code never changes meaning, whereas the message text is not contractual and may be reworded in any release
187
- - `bitmark` — source bitmark text
248
+ - `bitmark` — source bitmark text (a convenience echo; every bit is fully described by `bit` alone)
249
+
250
+ A region the parser could not read — an unknown bit type, or non-blank text
251
+ that never opened a bit — comes back as a bit of type `_error` beside its
252
+ `parser.errors` entry. It carries `originalType` (the header exactly as
253
+ written between the level dots and the `]`, including a leading `|` for a
254
+ commented-out bit and any `:format` / `&resource` suffix; absent when there
255
+ was no header) and `body` (the raw text after it, as a plain string), so
256
+ generating bitmark from it reproduces the original region.
188
257
 
189
258
  #### canonicalize(input: string, options?: CanonicalizeOptions): string
190
259
 
@@ -212,9 +281,15 @@ A patch document addresses bits by their position (or id) in the input
212
281
  {
213
282
  "version": 1,
214
283
  "entries": [
215
- { "select": { "index": 3, "id": "mc-014" }, "patch": [{ "path": "title", "op": "set", "value": "New" }] },
284
+ {
285
+ "select": { "index": 3, "id": "mc-014" },
286
+ "patch": [{ "path": "title", "op": "set", "value": "New" }]
287
+ },
216
288
  { "select": { "index": 9 }, "remove": true },
217
- { "insert": { "after": 4 }, "bit": { "type": "article", "body": "Inserted" } },
289
+ {
290
+ "insert": { "after": 4 },
291
+ "bit": { "type": "article", "body": "Inserted" }
292
+ },
218
293
  { "select": { "index": 7 }, "move": { "after": 12 } },
219
294
  { "select": "all", "patch": [{ "lang": "de" }] }
220
295
  ]
@@ -222,8 +297,10 @@ A patch document addresses bits by their position (or id) in the input
222
297
  ```
223
298
 
224
299
  `after` names an anchor — an input bit that is neither removed nor moved,
225
- or `-1` for the document start. `diff` produces such a document from two
226
- versions of a file (below).
300
+ or `-1` for the document start. A bit may be the target of at most one
301
+ `remove` or `move` entry: moving a bit twice, removing it twice, or removing
302
+ and moving it are all rejected, naming the second of the conflicting entries.
303
+ `diff` produces such a document from two versions of a file (below).
227
304
 
228
305
  #### diff(a: string, b: string, options?: DiffOptions): string
229
306
 
@@ -281,14 +358,74 @@ issue codes gained or lost, and the rendered `hunks`.
281
358
 
282
359
  Count or split the bits of a bitmark/JSON input without full parsing.
283
360
 
284
- #### lex(input: string, options?: LexOptions): string
285
-
286
- Lex bitmark text.
287
-
288
- Stages (`options.stage`):
289
-
290
- - `"lex"` — Unified lexer token stream, one token per line (default)
291
- - `"lex-json"` — Unified lexer token stream as JSON
361
+ #### semanticTokens(input: string, options?: SemanticTokensOptions): SemanticTokens
362
+
363
+ Parser-derived highlighting of bitmark source in the LSP semantic-tokens
364
+ shape — the typed form of
365
+ `convert(input, { inputFormat: "bitmark", outputFormat: "semantic-tokens" })`.
366
+ The parser runs (never the validator; diagnostics stay on their own channel)
367
+ and the AST is walked loss-lessly, so every character of the source is
368
+ covered exactly once and the highlighting agrees with the JSON the same
369
+ input produces: an unpaired `**` is text, an abandoned `[!tag` is text and
370
+ the `====` / `--` it would have swallowed are dividers, the body of a bit
371
+ whose body format is not bitmark (`[.code]`, `==== text ====`) is plain
372
+ text. Bitmark input only.
373
+
374
+ Options (`SemanticTokensOptions`, also `ConvertOptions` keys):
375
+
376
+ - `positionEncoding` — `"utf-16"` (default; the LSP default, what browser
377
+ editors index by) or `"utf-8"` (bytes). Echoed in the response.
378
+ - `tokensLayout` — `"lsp"` (default): `data`, the LSP relative encoding,
379
+ five integers per token (`deltaLine, deltaStart, length, tokenType,
380
+ tokenModifiers`; the type is a legend index, the modifiers a bitset); or
381
+ `"absolute"`: `tokens`, one `{ line, start, length, type, modifiers }`
382
+ per token, 0-based, by name. Both carry `legend` and `positionEncoding`.
383
+
384
+ Lines break at `\n`, `\r\n` and `\r`; no token contains a line terminator.
385
+ The legend orders are stable (a change is semver-major; new entries may be
386
+ appended). Token types name sigil classes and syntax positions, never a bit
387
+ or tag name; each has a recommended standard super type so an editor's
388
+ default theme colours it — paste the table into a VS Code extension's
389
+ `semanticTokenTypes`:
390
+
391
+ | type | covers | super type |
392
+ | ---------------------------------------------------------------------------------- | -------------------------------------------------------- | ---------------------------------- |
393
+ | `frontmatter` | the region before the first bit | `comment` |
394
+ | `bitSigil` | `[.`, level dots, `\|`, `:`, `&`, `]` of a bit header | `keyword` |
395
+ | `bitType` / `bitFormat` / `bitResourceType` | the header's three fields | `type` / `modifier` / `type` |
396
+ | `tagSigil` | every tag's own syntax (`[@`, `[!`, `[##`, `:`, `]`, …) | `keyword` |
397
+ | `propertyKey` / `resourceType` | a property tag's key / a resource tag's type | `property` / `type` |
398
+ | `tagText` | plain text inside any tag value | `string` |
399
+ | `cardDivider` / `sideDivider` / `variantDivider` / `footerDivider` / `textDivider` | `====`, `--`, `++`, `==== footer ====`, `==== text ====` | `operator` |
400
+ | `paragraphBreak` | a `\|` paragraph break | `operator` |
401
+ | `headingSigil` / `heading` | `# ` / the heading's plain text | `keyword` / `string` |
402
+ | `listMarker` | `• `, `•1 `, `•+ `, the indent | `keyword` |
403
+ | `codeSigil` / `codeLanguage` / `codeBody` | `\|code:` / the language / the body | `keyword` / `type` / `string` |
404
+ | `imageSigil` / `imageSrc` | `\|image:` / the source | `keyword` / `string` |
405
+ | `markSigil` | a mark's open and close markers | `operator` |
406
+ | `bold` / `italic` / `highlight` / `light` / `inline` | the marked text | `string` |
407
+ | `attrSigil` / `attrKey` / `attrValue` | an attr chain's `\|` and `:` / key / value | `operator` / `property` / `string` |
408
+ | `url` | a bare URL the parser auto-linked | `string` |
409
+ | `text` | plain body / card / footer text, and every top-level gap | (none) |
410
+ | `plainText` | a plain region (after `==== text ====`, or a raw body) | `string` |
411
+
412
+ Modifiers, in bit order: the tag classes `property`, `resource`, `title`,
413
+ `reference`, `anchor`, `item`, `instruction`, `hint`, `true`, `false`,
414
+ `gap`, `mark`, `solution` (every token of a tag of that class, sigils
415
+ included — exactly one per tag); `comment` (every token of a commented-out
416
+ bit); `unclosed` (tokens inside a parser unclosed-mark / unclosed-tag
417
+ notice). The full contract is `.zen/specs/API-SEM-semantic-tokens.tsp`.
418
+
419
+ #### The `lex` output format
420
+
421
+ `convert(input, { inputFormat: "bitmark", outputFormat: "lex" })` returns the
422
+ lexer's token stream as a JSON array: one `{ kind, span, text }` per token
423
+ (the `LexToken` type), `span` in bytes. Debug tooling — the token kinds are the
424
+ engine's own names, not a stable vocabulary, and the parser is never run
425
+ (pairing, abandoned-tag repair and raw-body handling all happen after lexing),
426
+ so a highlighter should use `semantic-tokens` instead. Bitmark input only; a
427
+ JSON or markup document has no source text to lex. The former `lex()` export
428
+ and `bitmark lex` command were removed in 7.0 (see the migration guide).
292
429
 
293
430
  #### breakscapeText(input: string, options?: BreakscapeOptions): string
294
431
 
@@ -325,13 +462,13 @@ version and (separately) any migration target.
325
462
  - `full` — complete detail for `"bit"`/`"all"` (default: `false` = compact view)
326
463
  - `language` — BCP-47 tag(s) for display names: one (`"de"`), several
327
464
  (`["de", "fr"]` or `"de,fr"`), or `"all"` — exported as `ALL_LANGUAGES`
328
- (default: English). More than one adds a `titles` map. See *Display names
329
- and languages* below
465
+ (default: English). More than one adds a `titles` map. See _Display names
466
+ and languages_ below
330
467
 
331
468
  `"bit-groups"` / `"resource-groups"` return the search/filter catalogs —
332
469
  each group's key, translated title, description, optional aliases and
333
470
  `subgroupOf`, and its members. They are available in EVERY build: the group
334
- catalog and per-bit titles/descriptions are "descriptive" data, which the
471
+ catalog and per-bit titles are "descriptive" data, which the
335
472
  lean wasm variants carry too. Deprecated members are excluded by default and
336
473
  MARKED when included with `includeDeprecated: true`; a search index matching
337
474
  already-published content needs them, because a migrating bit is re-emitted
@@ -341,13 +478,13 @@ To derive quiz categories, intersect a bit's `bitGroups` with the groups
341
478
  carrying `subgroupOf: "quizzes"` — `subgroupOf` is metadata and never implies
342
479
  membership, so every member of a subgroup also declares the parent.
343
480
 
344
- Only the META fields — TAG descriptions, group-inheritance provenance and raw
345
- mapping patterns (the `info-meta` cargo feature) — depend on the build. They
346
- are reported by the native CLI and the wasm `full` variant; `browser-full` and
347
- `bitmark-json` omit them to stay small. Those are absent structurally: the key
348
- is missing, never `null` or empty. Everything else `info` returns — including
349
- bit titles, bit descriptions and the group catalogs — is identical in every
350
- variant.
481
+ Only the META fields — BIT and TAG descriptions, group-inheritance provenance
482
+ and raw mapping patterns (the `info-meta` cargo feature) — depend on the build.
483
+ They are reported by the native CLI and the wasm `full` variant; `browser-full`
484
+ and `bitmark-json` omit them to stay small (the bit descriptions alone are
485
+ 14 KB gzip of the download). Those are absent structurally: the key is
486
+ missing, never `null` or empty. Everything else `info` returns — including bit
487
+ titles and the group catalogs — is identical in every variant.
351
488
 
352
489
  #### Display names and languages
353
490
 
@@ -390,7 +527,12 @@ which categories it is in":
390
527
 
391
528
  ```js
392
529
  const groups = JSON.parse(
393
- info({ infoType: "bit-groups", format: "json", includeDeprecated: true, language: "all" }),
530
+ info({
531
+ infoType: "bit-groups",
532
+ format: "json",
533
+ includeDeprecated: true,
534
+ language: "all",
535
+ }),
394
536
  );
395
537
  groups[0].bitTypes[0]; // { name: "assignment", title: "Assignment", titles: {…} }
396
538
  ```
@@ -508,7 +650,7 @@ const patch = patchEntry("id", "append", "1234");
508
650
 
509
651
  See `examples/` for runnable scripts covering the whole API (executed by
510
652
  `npm test`, so they stay correct). Generated API documentation (TypeDoc) is
511
- published at <https://getmorebrain.github.io/bitmark-parser-rust/docs/api/>
653
+ published at <https://getmorebrain.github.io/bitmark-parser/docs/api/>
512
654
  with each release (the site root hosts the bitmark language docs), or build
513
655
  it locally with `npm run docs` (→ `docs/api/`) and preview it with
514
656
  `npm run docs:serve` (http://localhost:8080, `--port` to change).
@@ -536,6 +678,28 @@ Note: deprecated bit types with a configured migration target (the
536
678
  `isCollapsible: true` — consumers keyed on the old `type` strings see the
537
679
  base type instead. `collapsible` itself is unaffected.
538
680
 
681
+ ### Property number values
682
+
683
+ A tag whose configured format is `number` accepts what JavaScript's
684
+ `Number(value)` accepts, minus the values that are not finite:
685
+
686
+ - surrounding whitespace, a leading `+`, leading zeros (`01`), a trailing
687
+ point (`1.`) and a leading point (`.5`);
688
+ - an exponent in either case (`1e2`, `1E2`);
689
+ - hexadecimal, binary and octal literals (`0x1F`, `0b101`, `0o701`) — prefix
690
+ case-insensitive, with no sign, fraction or exponent.
691
+
692
+ `NaN`, `Infinity`, an exponent beyond ±4000, and a radix literal past 64 bits
693
+ are **rejected**: the tag is dropped from the output and a
694
+ `property-format-mismatch` warning names it, exactly like any other value that
695
+ does not fit its format.
696
+
697
+ Accepted values print in JavaScript's `JSON.stringify` form — `0.5`, `1.5`,
698
+ `100`, `1e+21`, `1e-7` — so `[@width: 1.50]` emits `1.5` and `[@width:0x10]`
699
+ emits `16`. Values with more than 17 significant digits, including integers at
700
+ or above 2^63, keep the digits you wrote rather than being rounded through a
701
+ floating-point double.
702
+
539
703
  ## Legacy API (bpg-compatible)
540
704
 
541
705
  A compatibility facade for consumers migrating from
@@ -643,8 +807,12 @@ bitmark diff old.bitmark new.bitmark --output-format patch > d.patch
643
807
  bitmark transform old.bitmark --patch d.patch --output-format bitmark # reproduces new
644
808
  # --context N, --locate, --similarity F, --bbox-tolerance N, --color auto|always|never
645
809
 
646
- # Dump the lexer token stream
647
- bitmark lex input.bitmark
810
+ # Parser-derived highlighting: LSP semantic tokens (bitmark input only)
811
+ bitmark convert input.bitmark --input-format bitmark --output-format semantic-tokens
812
+ bitmark convert input.bitmark --output-format semantic-tokens --tokens-layout absolute --position-encoding utf-8 --pretty
813
+
814
+ # Dump the lexer token stream as JSON (debug tooling; bitmark input only)
815
+ bitmark convert input.bitmark --input-format bitmark --output-format lex --pretty
648
816
 
649
817
  # Breakscape / unbreakscape text
650
818
  bitmark breakscape input.txt
@@ -687,13 +855,13 @@ The package includes pre-built browser bundles with the WASM module.
687
855
  The CDN bundle carries all three variants' JS glue and fetches only the
688
856
  selected variant's `.wasm` (from `dist/browser/wasm/`). A bare `init()` loads
689
857
  `browser-full`; `init({ feature: "bitmark-json" })` fetches the smallest
690
- build, and a later `init({ feature: "full" })` upgrades in place (see *WASM
691
- variants*).
858
+ build, and a later `init({ feature: "full" })` upgrades in place (see _WASM
859
+ variants_).
692
860
 
693
861
  ### Bundler (webpack / vite)
694
862
 
695
863
  ```js
696
- import init, { convert, lex } from "@gmb/bitmark-parser/browser";
864
+ import init, { convert } from "@gmb/bitmark-parser/browser";
697
865
 
698
866
  await init();
699
867
  const json = convert("[.article]\nHello", { inputFormat: "bitmark" });