@gmb/bitmark-parser 6.11.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +311 -51
- package/dist/browser/bitmark-parser.min.js +4 -4
- package/dist/browser/bitmark-parser.min.js.map +1 -1
- package/dist/browser/cjs/index.cjs +696 -312
- package/dist/browser/cjs/index.cjs.map +1 -1
- package/dist/browser/cjs/index.d.cts +487 -74
- package/dist/browser/esm/index.d.ts +487 -74
- package/dist/browser/esm/index.js +693 -311
- package/dist/browser/esm/index.js.map +1 -1
- package/dist/browser/esm/worker-entry.js +578 -210
- package/dist/browser/esm/worker-entry.js.map +1 -1
- package/dist/browser/wasm/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_json_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_wasm_bg.wasm +0 -0
- package/dist/index.cjs +119 -107
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +487 -74
- package/dist/index.d.ts +487 -74
- package/dist/index.js +116 -106
- package/dist/index.js.map +1 -1
- package/dist/legacy.cjs +8 -30
- package/dist/legacy.cjs.map +1 -1
- package/dist/legacy.d.cts +0 -2
- package/dist/legacy.d.ts +0 -2
- package/dist/legacy.js +8 -30
- package/dist/legacy.js.map +1 -1
- package/dist/worker-entry.cjs +1 -5
- package/dist/worker-entry.cjs.map +1 -1
- package/package.json +7 -7
- package/schema/bitmark.schema.json +496 -139
- package/wasm/bitmark_wasm.d.ts +47 -20
- package/wasm/bitmark_wasm.js +262 -96
- package/wasm/bitmark_wasm_bg.wasm +0 -0
- package/wasm/bitmark_wasm_bg.wasm.d.ts +4 -2
- package/wasm/package.json +1 -1
- package/wasm-bitmark-json/bitmark_json_wasm.d.ts +26 -11
- package/wasm-bitmark-json/bitmark_json_wasm.js +195 -65
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm +0 -0
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm.d.ts +2 -1
- package/wasm-bitmark-json/package.json +1 -1
- package/wasm-browser-full/bitmark_browser_full_wasm.d.ts +47 -20
- package/wasm-browser-full/bitmark_browser_full_wasm.js +262 -96
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm.d.ts +4 -2
- package/wasm-browser-full/package.json +1 -1
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@ npm install @gmb/bitmark-parser
|
|
|
11
11
|
```js
|
|
12
12
|
import {
|
|
13
13
|
convert,
|
|
14
|
-
|
|
14
|
+
semanticTokens,
|
|
15
15
|
breakscapeText,
|
|
16
16
|
unbreakscapeText,
|
|
17
17
|
info,
|
|
@@ -32,11 +32,18 @@ console.log(parsedJson[0].bitmark);
|
|
|
32
32
|
// Generate bitmark from JSON
|
|
33
33
|
const bitmark = convert(json, { inputFormat: "json", outputFormat: "bitmark" });
|
|
34
34
|
|
|
35
|
-
//
|
|
36
|
-
const
|
|
35
|
+
// Parser-derived highlighting in the LSP semantic-tokens shape
|
|
36
|
+
const highlights = semanticTokens("[.article]\nHello **bold**", {
|
|
37
|
+
tokensLayout: "absolute",
|
|
38
|
+
});
|
|
37
39
|
|
|
38
|
-
//
|
|
39
|
-
const
|
|
40
|
+
// Dump the lexer's token stream (debug tooling; bitmark input only)
|
|
41
|
+
const tokens = JSON.parse(
|
|
42
|
+
convert("[.article]\nHello **bold**", {
|
|
43
|
+
inputFormat: "bitmark",
|
|
44
|
+
outputFormat: "lex",
|
|
45
|
+
}),
|
|
46
|
+
);
|
|
40
47
|
|
|
41
48
|
// Breakscape / unbreakscape text
|
|
42
49
|
const escaped = breakscapeText("[!example]", {
|
|
@@ -60,23 +67,37 @@ console.log(version());
|
|
|
60
67
|
The package ships **three wasm builds of the same engine** and selects one at
|
|
61
68
|
runtime — the API surface never changes:
|
|
62
69
|
|
|
63
|
-
| Feature
|
|
64
|
-
|
|
|
65
|
-
| `full` (Node default)
|
|
66
|
-
| `browser-full` (browser default) | the same conversions
|
|
67
|
-
| `bitmark-json`
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
70
|
+
| Feature | Contents | Size (approx.) |
|
|
71
|
+
| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------- |
|
|
72
|
+
| `full` (Node default) | everything: rich `info` metadata + built-in translations + the semantic `diff` | ~990 KB (~378 KB gzip) |
|
|
73
|
+
| `browser-full` (browser default) | the same conversions and `diff`, without either | ~783 KB (~317 KB gzip) |
|
|
74
|
+
| `bitmark-json` | bitmark ↔ JSON only: `convert`/`canonicalize`/`transform` (formats `auto`/`bitmark`/`json`, plus the `text` and `semantic-tokens` outputs), `info` (JSON only), breakscape, text fragments — no `diff` | ~563 KB (~224 KB gzip) |
|
|
75
|
+
|
|
76
|
+
**The rule.** _Every build includes every feature, except the two
|
|
77
|
+
browser-targeted variants, which each carry an explicit exclusion list._ That
|
|
78
|
+
is the whole story — a new capability is in every build unless it is named
|
|
79
|
+
below:
|
|
80
|
+
|
|
81
|
+
| build | features |
|
|
82
|
+
| -------------------------------------------------- | ----------------------------------------------------------------------------------- |
|
|
83
|
+
| native CLI, daemon, docgen, schemagen, wasm `full` | **all** |
|
|
84
|
+
| wasm `browser-full` | all **except** `info-meta` (bit + tag prose) and `translations` (the baked table) |
|
|
85
|
+
| wasm `bitmark-json` | all **except** the above and `markup`, `mapping-report`, `diff`, `lex`, `info-text` |
|
|
86
|
+
|
|
87
|
+
So, not in `bitmark-json`: the markup formats (`html`, `xml`,
|
|
88
|
+
`xml-niso-iec`, …) as input or output (they throw an `unsupported` error), the
|
|
89
|
+
`lex` output format and the `mappingReport` option (same error), `diff`
|
|
90
|
+
(throws `UnsupportedFeatureError`), and the human-readable rendering of `info` —
|
|
91
|
+
`info({ format: "text" })` throws `UnsupportedFeatureError` there, so pass
|
|
92
|
+
`format: "json"` (the JSON views are the contractual ones and are in every
|
|
93
|
+
variant).
|
|
73
94
|
|
|
74
95
|
`full` and `browser-full` convert **identically** — they differ only in what
|
|
75
|
-
`info` can report: the META fields (tag descriptions, group provenance
|
|
76
|
-
mapping patterns) and the built-in translations table. Bit titles
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
96
|
+
`info` can report: the META fields (bit and tag descriptions, group provenance
|
|
97
|
+
and raw mapping patterns) and the built-in translations table. Bit titles and
|
|
98
|
+
the bit-group / resource-group catalogs are in every variant, in English;
|
|
99
|
+
`register` supplies translations to the lean ones (see _Display names and
|
|
100
|
+
languages_).
|
|
80
101
|
|
|
81
102
|
Why the defaults differ: only the browser pays a download-latency cost for
|
|
82
103
|
those strings, and a Node backend loading wasm from disk has the same
|
|
@@ -89,15 +110,15 @@ browser, select `browser-full` explicitly.**
|
|
|
89
110
|
> **Migrating from ≤ 6.10:** `full` used to mean what `browser-full` means now.
|
|
90
111
|
> Browser code that explicitly selected `"full"` should move to
|
|
91
112
|
> `"browser-full"` to keep its download size close to what it was; leave it on
|
|
92
|
-
> `"full"` to gain the metadata and built-in translations for ~
|
|
113
|
+
> `"full"` to gain the metadata and built-in translations for ~62 KB gzip.
|
|
93
114
|
> Nothing else changes — the capabilities that were in `full` are in both.
|
|
94
115
|
|
|
95
116
|
```js
|
|
96
117
|
import { init, convert, variant } from "@gmb/bitmark-parser";
|
|
97
118
|
|
|
98
119
|
await init({ feature: "bitmark-json" }); // load the minimal wasm
|
|
99
|
-
convert("[.article]\nHello");
|
|
100
|
-
variant();
|
|
120
|
+
convert("[.article]\nHello"); // works
|
|
121
|
+
variant(); // "bitmark-json"
|
|
101
122
|
|
|
102
123
|
// Upgrade in place: the interface stays valid while the full wasm loads —
|
|
103
124
|
// calls keep being served by the active module, then swap atomically.
|
|
@@ -130,6 +151,39 @@ Notes:
|
|
|
130
151
|
|
|
131
152
|
### API
|
|
132
153
|
|
|
154
|
+
#### Errors
|
|
155
|
+
|
|
156
|
+
**Errors are thrown, never returned.** Every fallible WASM export raises a
|
|
157
|
+
JavaScript `Error` whose `message` has one shape:
|
|
158
|
+
|
|
159
|
+
```text
|
|
160
|
+
<kind> at <path>: <message>
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
A returned string is therefore always a result — there is no prefix or
|
|
164
|
+
sentinel to test for, and every call site is a `try`/`catch`.
|
|
165
|
+
|
|
166
|
+
`<kind>` is a **stable kebab-case identifier** — the part to match on.
|
|
167
|
+
`<path>` locates the failure (a JSON pointer-ish path, `offset N` for a JSON
|
|
168
|
+
syntax error, or empty) and `<message>` is human-readable prose that may
|
|
169
|
+
change in any release. The kinds:
|
|
170
|
+
|
|
171
|
+
| kind | meaning |
|
|
172
|
+
| -------------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
|
|
173
|
+
| `invalid-json` | the input string is not well-formed JSON |
|
|
174
|
+
| `compose` | the reverse composer rejected the JSON (shape / field / type) |
|
|
175
|
+
| `unsupported` | the requested conversion is not available (an output-only format used as input, or a format compiled out of this variant) |
|
|
176
|
+
| `patch-parse-error` | a patch or patch document is malformed |
|
|
177
|
+
| `patch-index-out-of-range` | a patch path addressed an array element that does not exist |
|
|
178
|
+
| `patch-type-mismatch` | a patch value's JSON type is incompatible with its target |
|
|
179
|
+
| `patch-conflict` | a patch document's entries contradict each other (a bit removed or moved twice, or an anchor that is itself removed or moved) |
|
|
180
|
+
|
|
181
|
+
Every TypeScript wrapper propagates that `Error` unchanged — `convert`,
|
|
182
|
+
`transform`, `canonicalize`, `bitmarkToObjects`, `countBits`, `diff`, … — as
|
|
183
|
+
does the parallel worker pool, which rejects with the first failing bit's
|
|
184
|
+
error. `register` re-wraps it as a `RegisterError`; a capability the loaded
|
|
185
|
+
variant compiled out raises `UnsupportedFeatureError` before any wasm call.
|
|
186
|
+
|
|
133
187
|
#### convert(input: string, options?: ConvertOptions): string
|
|
134
188
|
|
|
135
189
|
Convert between bitmark, JSON, and the mapped markup formats (html/xml/…).
|
|
@@ -153,7 +207,7 @@ Options:
|
|
|
153
207
|
the bit is `_`-prefixed. A bare tag the bit does not declare (`[!…]`,
|
|
154
208
|
`[?…]`, `[#…]`, … on a bit whose config has no such tag) is an unknown
|
|
155
209
|
too, under the key a declaring bit would use (`[!x]` → `"instruction":
|
|
156
|
-
|
|
210
|
+
["x"]`); a rejected resource attachment is not — it is still emitted, with
|
|
157
211
|
a warning. Unknown properties are never converted back to bitmark, and
|
|
158
212
|
each occurrence warns whether or not it is included.
|
|
159
213
|
Also available on `bitmarkToObjects` and as the CLI's
|
|
@@ -175,6 +229,13 @@ Options:
|
|
|
175
229
|
> format, set `inputFormat` explicitly. The same applies to `canonicalize`
|
|
176
230
|
> and `transform`, whose `inputFormat` also defaults to `"auto"`.
|
|
177
231
|
|
|
232
|
+
> **Input limits:** JSON objects/arrays and markup elements may nest at most
|
|
233
|
+
> **256** levels deep. Deeper JSON is rejected like any other malformed JSON
|
|
234
|
+
> (`"error: … nesting depth exceeds 256"`); a deeper markup open tag is kept
|
|
235
|
+
> as literal text. The bound keeps a crafted document from exhausting the
|
|
236
|
+
> WASM stack — an instance that rejects a too-deep input stays usable for the
|
|
237
|
+
> next call.
|
|
238
|
+
|
|
178
239
|
`convert` is the single string-based conversion surface — parsing is
|
|
179
240
|
`{ inputFormat: "bitmark", outputFormat: "json" }`, generating is
|
|
180
241
|
`{ inputFormat: "json", outputFormat: "bitmark" }`. For typed objects use
|
|
@@ -184,7 +245,15 @@ For bitmark input, output is a JSON array of entries with:
|
|
|
184
245
|
|
|
185
246
|
- `bit` — serialized bit payload
|
|
186
247
|
- `parser` — parser metadata, `errors` / `warnings`, and `infos` (informational notices for recoveries that are not necessarily wrong, e.g. text kept as text because it only looks like an unclosed tag). Each issue carries a stable machine-readable `code` (`"unknown-property"`, `"missing-required-tag"`, …) beside its human-readable `message`, plus `text` and `location`. **Branch on `code`** — a shipped code never changes meaning, whereas the message text is not contractual and may be reworded in any release
|
|
187
|
-
- `bitmark` — source bitmark text
|
|
248
|
+
- `bitmark` — source bitmark text (a convenience echo; every bit is fully described by `bit` alone)
|
|
249
|
+
|
|
250
|
+
A region the parser could not read — an unknown bit type, or non-blank text
|
|
251
|
+
that never opened a bit — comes back as a bit of type `_error` beside its
|
|
252
|
+
`parser.errors` entry. It carries `originalType` (the header exactly as
|
|
253
|
+
written between the level dots and the `]`, including a leading `|` for a
|
|
254
|
+
commented-out bit and any `:format` / `&resource` suffix; absent when there
|
|
255
|
+
was no header) and `body` (the raw text after it, as a plain string), so
|
|
256
|
+
generating bitmark from it reproduces the original region.
|
|
188
257
|
|
|
189
258
|
#### canonicalize(input: string, options?: CanonicalizeOptions): string
|
|
190
259
|
|
|
@@ -192,20 +261,171 @@ Re-emit input in its own format in canonical form — the single same-format sur
|
|
|
192
261
|
|
|
193
262
|
#### transform(input: string, options?: TransformOptions): string
|
|
194
263
|
|
|
195
|
-
Apply
|
|
264
|
+
Apply a patch document to a document, bit by bit, and re-emit it.
|
|
265
|
+
`transformParallel` is the worker-thread variant returning a `Promise<string>`.
|
|
266
|
+
|
|
267
|
+
- `patch` — a `PatchDocument`, or the shorthand: a bare array of entries
|
|
268
|
+
applied to every bit (build entries with `patchEntry`). Either form may be
|
|
269
|
+
a JSON string
|
|
270
|
+
- `preHook` / `postHook` — per-bit hooks; the pre-hook may return patches, an
|
|
271
|
+
output-format override or `drop`. Hooks see the original bits, in order,
|
|
272
|
+
before the document reshapes anything; a hook's patches apply after the
|
|
273
|
+
document's
|
|
274
|
+
- `inputFormat`, `outputFormat`, `mode`, `pretty`, `indent`,
|
|
275
|
+
`spacesAroundValues` — as for `convert`
|
|
276
|
+
|
|
277
|
+
A patch document addresses bits by their position (or id) in the input
|
|
278
|
+
**before** any entry is applied, so it reads like a diff:
|
|
279
|
+
|
|
280
|
+
```json
|
|
281
|
+
{
|
|
282
|
+
"version": 1,
|
|
283
|
+
"entries": [
|
|
284
|
+
{
|
|
285
|
+
"select": { "index": 3, "id": "mc-014" },
|
|
286
|
+
"patch": [{ "path": "title", "op": "set", "value": "New" }]
|
|
287
|
+
},
|
|
288
|
+
{ "select": { "index": 9 }, "remove": true },
|
|
289
|
+
{
|
|
290
|
+
"insert": { "after": 4 },
|
|
291
|
+
"bit": { "type": "article", "body": "Inserted" }
|
|
292
|
+
},
|
|
293
|
+
{ "select": { "index": 7 }, "move": { "after": 12 } },
|
|
294
|
+
{ "select": "all", "patch": [{ "lang": "de" }] }
|
|
295
|
+
]
|
|
296
|
+
}
|
|
297
|
+
```
|
|
196
298
|
|
|
197
|
-
|
|
299
|
+
`after` names an anchor — an input bit that is neither removed nor moved,
|
|
300
|
+
or `-1` for the document start. A bit may be the target of at most one
|
|
301
|
+
`remove` or `move` entry: moving a bit twice, removing it twice, or removing
|
|
302
|
+
and moving it are all rejected, naming the second of the conflicting entries.
|
|
303
|
+
`diff` produces such a document from two versions of a file (below).
|
|
198
304
|
|
|
199
|
-
|
|
305
|
+
#### diff(a: string, b: string, options?: DiffOptions): string
|
|
200
306
|
|
|
201
|
-
|
|
307
|
+
Semantic diff of two documents. Both sides are re-canonicalized and
|
|
308
|
+
compared as bit JSON, so tag order, spacing, breakscaping and omitted
|
|
309
|
+
defaults are not differences; every real change is rendered as the
|
|
310
|
+
canonical bitmark an author would type. The inputs may be bitmark or JSON,
|
|
311
|
+
and need not share a format.
|
|
202
312
|
|
|
203
|
-
|
|
313
|
+
```
|
|
314
|
+
@@ bit 1 -> 2 (id mc-014) [.multiple-choice] changed
|
|
315
|
+
instruction[0]
|
|
316
|
+
-[!Pick one]
|
|
317
|
+
+[!Pick exactly one]
|
|
318
|
+
quizzes[0]
|
|
319
|
+
====
|
|
320
|
+
[-Bonn]
|
|
321
|
+
-[-Berlin]
|
|
322
|
+
+[+Berlin]
|
|
323
|
+
@@ bit 4 (id hero) [.image] removed
|
|
324
|
+
-[.image]
|
|
325
|
+
-[@id:hero]
|
|
326
|
+
-[&image:https://example.com/hero.png]
|
|
327
|
+
```
|
|
204
328
|
|
|
205
|
-
|
|
329
|
+
Bits are paired by `id`, then by `sourceBB` on the same `sourceRL`, then by
|
|
330
|
+
exact content, then by text similarity; a bit whose position changed among
|
|
331
|
+
the kept ones is reported `moved`. Empty output means the documents are
|
|
332
|
+
semantically equal.
|
|
333
|
+
|
|
334
|
+
- `outputFormat` — `"bitmark"` (default; the text above), `"json"` (one
|
|
335
|
+
object per aligned bit, see `diffBits`), or `"patch"` (a patch document
|
|
336
|
+
that turns `a` into `b` — feed it to `transform`)
|
|
337
|
+
- `inputFormat` — applies to both sides (default: `"auto"`, sniffed per side)
|
|
338
|
+
- `context` — unchanged lines kept around each change (default: `3`)
|
|
339
|
+
- `locate` — add source line numbers to bit positions (bitmark inputs; default: `false`)
|
|
340
|
+
- `similarity` — pairing threshold for bits with no id, box or exact match, `0..1` (default: `0.6`; `1` disables)
|
|
341
|
+
- `bboxTolerance` — pixels per coordinate for the `sourceBB` stage (default: `8`)
|
|
342
|
+
- `spacesAroundValues` — in the rendered bitmark lines (default: `0`)
|
|
343
|
+
|
|
344
|
+
The text layout and the JSON shape are stable contracts. An unknown
|
|
345
|
+
property is shown as a JSON line, since the generator cannot render it, and
|
|
346
|
+
cannot be re-applied. Not available on the `bitmark-json` variant
|
|
347
|
+
(`UnsupportedFeatureError`).
|
|
348
|
+
|
|
349
|
+
#### diffBits(a: string, b: string, options?: DiffOptions): BitDiff[]
|
|
350
|
+
|
|
351
|
+
The typed form of `diff`: one `BitDiff` per aligned bit, unchanged ones
|
|
352
|
+
included so the alignment is complete — `status`
|
|
353
|
+
(`unchanged | changed | moved | removed | added`), the `a`/`b` positions,
|
|
354
|
+
`ops` (a patch turning the A bit into the B bit) and `revert`, the parser
|
|
355
|
+
issue codes gained or lost, and the rendered `hunks`.
|
|
206
356
|
|
|
207
|
-
|
|
208
|
-
|
|
357
|
+
#### countBits(input: string): number / splitBits(input: string): BitSlice[]
|
|
358
|
+
|
|
359
|
+
Count or split the bits of a bitmark/JSON input without full parsing.
|
|
360
|
+
|
|
361
|
+
#### semanticTokens(input: string, options?: SemanticTokensOptions): SemanticTokens
|
|
362
|
+
|
|
363
|
+
Parser-derived highlighting of bitmark source in the LSP semantic-tokens
|
|
364
|
+
shape — the typed form of
|
|
365
|
+
`convert(input, { inputFormat: "bitmark", outputFormat: "semantic-tokens" })`.
|
|
366
|
+
The parser runs (never the validator; diagnostics stay on their own channel)
|
|
367
|
+
and the AST is walked loss-lessly, so every character of the source is
|
|
368
|
+
covered exactly once and the highlighting agrees with the JSON the same
|
|
369
|
+
input produces: an unpaired `**` is text, an abandoned `[!tag` is text and
|
|
370
|
+
the `====` / `--` it would have swallowed are dividers, the body of a bit
|
|
371
|
+
whose body format is not bitmark (`[.code]`, `==== text ====`) is plain
|
|
372
|
+
text. Bitmark input only.
|
|
373
|
+
|
|
374
|
+
Options (`SemanticTokensOptions`, also `ConvertOptions` keys):
|
|
375
|
+
|
|
376
|
+
- `positionEncoding` — `"utf-16"` (default; the LSP default, what browser
|
|
377
|
+
editors index by) or `"utf-8"` (bytes). Echoed in the response.
|
|
378
|
+
- `tokensLayout` — `"lsp"` (default): `data`, the LSP relative encoding,
|
|
379
|
+
five integers per token (`deltaLine, deltaStart, length, tokenType,
|
|
380
|
+
tokenModifiers`; the type is a legend index, the modifiers a bitset); or
|
|
381
|
+
`"absolute"`: `tokens`, one `{ line, start, length, type, modifiers }`
|
|
382
|
+
per token, 0-based, by name. Both carry `legend` and `positionEncoding`.
|
|
383
|
+
|
|
384
|
+
Lines break at `\n`, `\r\n` and `\r`; no token contains a line terminator.
|
|
385
|
+
The legend orders are stable (a change is semver-major; new entries may be
|
|
386
|
+
appended). Token types name sigil classes and syntax positions, never a bit
|
|
387
|
+
or tag name; each has a recommended standard super type so an editor's
|
|
388
|
+
default theme colours it — paste the table into a VS Code extension's
|
|
389
|
+
`semanticTokenTypes`:
|
|
390
|
+
|
|
391
|
+
| type | covers | super type |
|
|
392
|
+
| ---------------------------------------------------------------------------------- | -------------------------------------------------------- | ---------------------------------- |
|
|
393
|
+
| `frontmatter` | the region before the first bit | `comment` |
|
|
394
|
+
| `bitSigil` | `[.`, level dots, `\|`, `:`, `&`, `]` of a bit header | `keyword` |
|
|
395
|
+
| `bitType` / `bitFormat` / `bitResourceType` | the header's three fields | `type` / `modifier` / `type` |
|
|
396
|
+
| `tagSigil` | every tag's own syntax (`[@`, `[!`, `[##`, `:`, `]`, …) | `keyword` |
|
|
397
|
+
| `propertyKey` / `resourceType` | a property tag's key / a resource tag's type | `property` / `type` |
|
|
398
|
+
| `tagText` | plain text inside any tag value | `string` |
|
|
399
|
+
| `cardDivider` / `sideDivider` / `variantDivider` / `footerDivider` / `textDivider` | `====`, `--`, `++`, `==== footer ====`, `==== text ====` | `operator` |
|
|
400
|
+
| `paragraphBreak` | a `\|` paragraph break | `operator` |
|
|
401
|
+
| `headingSigil` / `heading` | `# ` / the heading's plain text | `keyword` / `string` |
|
|
402
|
+
| `listMarker` | `• `, `•1 `, `•+ `, the indent | `keyword` |
|
|
403
|
+
| `codeSigil` / `codeLanguage` / `codeBody` | `\|code:` / the language / the body | `keyword` / `type` / `string` |
|
|
404
|
+
| `imageSigil` / `imageSrc` | `\|image:` / the source | `keyword` / `string` |
|
|
405
|
+
| `markSigil` | a mark's open and close markers | `operator` |
|
|
406
|
+
| `bold` / `italic` / `highlight` / `light` / `inline` | the marked text | `string` |
|
|
407
|
+
| `attrSigil` / `attrKey` / `attrValue` | an attr chain's `\|` and `:` / key / value | `operator` / `property` / `string` |
|
|
408
|
+
| `url` | a bare URL the parser auto-linked | `string` |
|
|
409
|
+
| `text` | plain body / card / footer text, and every top-level gap | (none) |
|
|
410
|
+
| `plainText` | a plain region (after `==== text ====`, or a raw body) | `string` |
|
|
411
|
+
|
|
412
|
+
Modifiers, in bit order: the tag classes `property`, `resource`, `title`,
|
|
413
|
+
`reference`, `anchor`, `item`, `instruction`, `hint`, `true`, `false`,
|
|
414
|
+
`gap`, `mark`, `solution` (every token of a tag of that class, sigils
|
|
415
|
+
included — exactly one per tag); `comment` (every token of a commented-out
|
|
416
|
+
bit); `unclosed` (tokens inside a parser unclosed-mark / unclosed-tag
|
|
417
|
+
notice). The full contract is `.zen/specs/API-SEM-semantic-tokens.tsp`.
|
|
418
|
+
|
|
419
|
+
#### The `lex` output format
|
|
420
|
+
|
|
421
|
+
`convert(input, { inputFormat: "bitmark", outputFormat: "lex" })` returns the
|
|
422
|
+
lexer's token stream as a JSON array: one `{ kind, span, text }` per token
|
|
423
|
+
(the `LexToken` type), `span` in bytes. Debug tooling — the token kinds are the
|
|
424
|
+
engine's own names, not a stable vocabulary, and the parser is never run
|
|
425
|
+
(pairing, abandoned-tag repair and raw-body handling all happen after lexing),
|
|
426
|
+
so a highlighter should use `semantic-tokens` instead. Bitmark input only; a
|
|
427
|
+
JSON or markup document has no source text to lex. The former `lex()` export
|
|
428
|
+
and `bitmark lex` command were removed in 7.0 (see the migration guide).
|
|
209
429
|
|
|
210
430
|
#### breakscapeText(input: string, options?: BreakscapeOptions): string
|
|
211
431
|
|
|
@@ -242,13 +462,13 @@ version and (separately) any migration target.
|
|
|
242
462
|
- `full` — complete detail for `"bit"`/`"all"` (default: `false` = compact view)
|
|
243
463
|
- `language` — BCP-47 tag(s) for display names: one (`"de"`), several
|
|
244
464
|
(`["de", "fr"]` or `"de,fr"`), or `"all"` — exported as `ALL_LANGUAGES`
|
|
245
|
-
(default: English). More than one adds a `titles` map. See
|
|
246
|
-
and
|
|
465
|
+
(default: English). More than one adds a `titles` map. See _Display names
|
|
466
|
+
and languages_ below
|
|
247
467
|
|
|
248
468
|
`"bit-groups"` / `"resource-groups"` return the search/filter catalogs —
|
|
249
469
|
each group's key, translated title, description, optional aliases and
|
|
250
470
|
`subgroupOf`, and its members. They are available in EVERY build: the group
|
|
251
|
-
catalog and per-bit titles
|
|
471
|
+
catalog and per-bit titles are "descriptive" data, which the
|
|
252
472
|
lean wasm variants carry too. Deprecated members are excluded by default and
|
|
253
473
|
MARKED when included with `includeDeprecated: true`; a search index matching
|
|
254
474
|
already-published content needs them, because a migrating bit is re-emitted
|
|
@@ -258,13 +478,13 @@ To derive quiz categories, intersect a bit's `bitGroups` with the groups
|
|
|
258
478
|
carrying `subgroupOf: "quizzes"` — `subgroupOf` is metadata and never implies
|
|
259
479
|
membership, so every member of a subgroup also declares the parent.
|
|
260
480
|
|
|
261
|
-
Only the META fields — TAG descriptions, group-inheritance provenance
|
|
262
|
-
mapping patterns (the `info-meta` cargo feature) — depend on the build.
|
|
263
|
-
are reported by the native CLI and the wasm `full` variant; `browser-full`
|
|
264
|
-
`bitmark-json` omit them to stay small
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
variant.
|
|
481
|
+
Only the META fields — BIT and TAG descriptions, group-inheritance provenance
|
|
482
|
+
and raw mapping patterns (the `info-meta` cargo feature) — depend on the build.
|
|
483
|
+
They are reported by the native CLI and the wasm `full` variant; `browser-full`
|
|
484
|
+
and `bitmark-json` omit them to stay small (the bit descriptions alone are
|
|
485
|
+
14 KB gzip of the download). Those are absent structurally: the key is
|
|
486
|
+
missing, never `null` or empty. Everything else `info` returns — including bit
|
|
487
|
+
titles and the group catalogs — is identical in every variant.
|
|
268
488
|
|
|
269
489
|
#### Display names and languages
|
|
270
490
|
|
|
@@ -307,7 +527,12 @@ which categories it is in":
|
|
|
307
527
|
|
|
308
528
|
```js
|
|
309
529
|
const groups = JSON.parse(
|
|
310
|
-
info({
|
|
530
|
+
info({
|
|
531
|
+
infoType: "bit-groups",
|
|
532
|
+
format: "json",
|
|
533
|
+
includeDeprecated: true,
|
|
534
|
+
language: "all",
|
|
535
|
+
}),
|
|
311
536
|
);
|
|
312
537
|
groups[0].bitTypes[0]; // { name: "assignment", title: "Assignment", titles: {…} }
|
|
313
538
|
```
|
|
@@ -425,7 +650,7 @@ const patch = patchEntry("id", "append", "1234");
|
|
|
425
650
|
|
|
426
651
|
See `examples/` for runnable scripts covering the whole API (executed by
|
|
427
652
|
`npm test`, so they stay correct). Generated API documentation (TypeDoc) is
|
|
428
|
-
published at <https://getmorebrain.github.io/bitmark-parser
|
|
653
|
+
published at <https://getmorebrain.github.io/bitmark-parser/docs/api/>
|
|
429
654
|
with each release (the site root hosts the bitmark language docs), or build
|
|
430
655
|
it locally with `npm run docs` (→ `docs/api/`) and preview it with
|
|
431
656
|
`npm run docs:serve` (http://localhost:8080, `--port` to change).
|
|
@@ -453,6 +678,28 @@ Note: deprecated bit types with a configured migration target (the
|
|
|
453
678
|
`isCollapsible: true` — consumers keyed on the old `type` strings see the
|
|
454
679
|
base type instead. `collapsible` itself is unaffected.
|
|
455
680
|
|
|
681
|
+
### Property number values
|
|
682
|
+
|
|
683
|
+
A tag whose configured format is `number` accepts what JavaScript's
|
|
684
|
+
`Number(value)` accepts, minus the values that are not finite:
|
|
685
|
+
|
|
686
|
+
- surrounding whitespace, a leading `+`, leading zeros (`01`), a trailing
|
|
687
|
+
point (`1.`) and a leading point (`.5`);
|
|
688
|
+
- an exponent in either case (`1e2`, `1E2`);
|
|
689
|
+
- hexadecimal, binary and octal literals (`0x1F`, `0b101`, `0o701`) — prefix
|
|
690
|
+
case-insensitive, with no sign, fraction or exponent.
|
|
691
|
+
|
|
692
|
+
`NaN`, `Infinity`, an exponent beyond ±4000, and a radix literal past 64 bits
|
|
693
|
+
are **rejected**: the tag is dropped from the output and a
|
|
694
|
+
`property-format-mismatch` warning names it, exactly like any other value that
|
|
695
|
+
does not fit its format.
|
|
696
|
+
|
|
697
|
+
Accepted values print in JavaScript's `JSON.stringify` form — `0.5`, `1.5`,
|
|
698
|
+
`100`, `1e+21`, `1e-7` — so `[@width: 1.50]` emits `1.5` and `[@width:0x10]`
|
|
699
|
+
emits `16`. Values with more than 17 significant digits, including integers at
|
|
700
|
+
or above 2^63, keep the digits you wrote rather than being rounded through a
|
|
701
|
+
floating-point double.
|
|
702
|
+
|
|
456
703
|
## Legacy API (bpg-compatible)
|
|
457
704
|
|
|
458
705
|
A compatibility facade for consumers migrating from
|
|
@@ -548,11 +795,24 @@ bitmark convert standard.xml --input-format xml-niso-iec --output-format bitmark
|
|
|
548
795
|
bitmark canonicalize input.bitmark
|
|
549
796
|
bitmark canonicalize input.json --mode full
|
|
550
797
|
|
|
551
|
-
# Apply
|
|
798
|
+
# Apply a patch document (per-bit patches by index/id, remove, insert, move;
|
|
799
|
+
# a bare array of entries still patches every bit)
|
|
552
800
|
bitmark transform input.json --patch patches.json # alias: patch
|
|
553
801
|
|
|
554
|
-
#
|
|
555
|
-
bitmark
|
|
802
|
+
# Semantic diff of two documents, rendered as bitmark
|
|
803
|
+
bitmark diff old.bitmark new.bitmark # unified-style text
|
|
804
|
+
bitmark diff old.bitmark new.bitmark --stat --exit-code # counts; exit 1 when they differ
|
|
805
|
+
bitmark diff old.bitmark new.json --output-format json # per-bit ops, revert, hunks
|
|
806
|
+
bitmark diff old.bitmark new.bitmark --output-format patch > d.patch
|
|
807
|
+
bitmark transform old.bitmark --patch d.patch --output-format bitmark # reproduces new
|
|
808
|
+
# --context N, --locate, --similarity F, --bbox-tolerance N, --color auto|always|never
|
|
809
|
+
|
|
810
|
+
# Parser-derived highlighting: LSP semantic tokens (bitmark input only)
|
|
811
|
+
bitmark convert input.bitmark --input-format bitmark --output-format semantic-tokens
|
|
812
|
+
bitmark convert input.bitmark --output-format semantic-tokens --tokens-layout absolute --position-encoding utf-8 --pretty
|
|
813
|
+
|
|
814
|
+
# Dump the lexer token stream as JSON (debug tooling; bitmark input only)
|
|
815
|
+
bitmark convert input.bitmark --input-format bitmark --output-format lex --pretty
|
|
556
816
|
|
|
557
817
|
# Breakscape / unbreakscape text
|
|
558
818
|
bitmark breakscape input.txt
|
|
@@ -595,13 +855,13 @@ The package includes pre-built browser bundles with the WASM module.
|
|
|
595
855
|
The CDN bundle carries all three variants' JS glue and fetches only the
|
|
596
856
|
selected variant's `.wasm` (from `dist/browser/wasm/`). A bare `init()` loads
|
|
597
857
|
`browser-full`; `init({ feature: "bitmark-json" })` fetches the smallest
|
|
598
|
-
build, and a later `init({ feature: "full" })` upgrades in place (see
|
|
599
|
-
|
|
858
|
+
build, and a later `init({ feature: "full" })` upgrades in place (see _WASM
|
|
859
|
+
variants_).
|
|
600
860
|
|
|
601
861
|
### Bundler (webpack / vite)
|
|
602
862
|
|
|
603
863
|
```js
|
|
604
|
-
import init, { convert
|
|
864
|
+
import init, { convert } from "@gmb/bitmark-parser/browser";
|
|
605
865
|
|
|
606
866
|
await init();
|
|
607
867
|
const json = convert("[.article]\nHello", { inputFormat: "bitmark" });
|