@gmb/bitmark-parser 6.11.1 → 7.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +361 -57
- package/dist/browser/bitmark-parser.min.js +4 -4
- package/dist/browser/bitmark-parser.min.js.map +1 -1
- package/dist/browser/cjs/index.cjs +966 -372
- package/dist/browser/cjs/index.cjs.map +1 -1
- package/dist/browser/cjs/index.d.cts +642 -70
- package/dist/browser/esm/index.d.ts +642 -70
- package/dist/browser/esm/index.js +958 -371
- package/dist/browser/esm/index.js.map +1 -1
- package/dist/browser/esm/worker-entry.js +790 -247
- package/dist/browser/esm/worker-entry.js.map +1 -1
- package/dist/browser/wasm/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_json_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_wasm_bg.wasm +0 -0
- package/dist/index.cjs +163 -124
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +637 -70
- package/dist/index.d.ts +637 -70
- package/dist/index.js +155 -123
- package/dist/index.js.map +1 -1
- package/dist/legacy.cjs +8 -30
- package/dist/legacy.cjs.map +1 -1
- package/dist/legacy.d.cts +0 -2
- package/dist/legacy.d.ts +0 -2
- package/dist/legacy.js +8 -30
- package/dist/legacy.js.map +1 -1
- package/dist/worker-entry.cjs +1 -5
- package/dist/worker-entry.cjs.map +1 -1
- package/package.json +7 -7
- package/schema/bitmark.schema.json +507 -141
- package/wasm/bitmark_wasm.d.ts +65 -26
- package/wasm/bitmark_wasm.js +348 -116
- package/wasm/bitmark_wasm_bg.wasm +0 -0
- package/wasm/bitmark_wasm_bg.wasm.d.ts +6 -3
- package/wasm/package.json +1 -1
- package/wasm-bitmark-json/bitmark_json_wasm.d.ts +56 -15
- package/wasm-bitmark-json/bitmark_json_wasm.js +321 -77
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm +0 -0
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm.d.ts +5 -2
- package/wasm-bitmark-json/package.json +1 -1
- package/wasm-browser-full/bitmark_browser_full_wasm.d.ts +65 -26
- package/wasm-browser-full/bitmark_browser_full_wasm.js +348 -116
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm.d.ts +6 -3
- package/wasm-browser-full/package.json +1 -1
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@ npm install @gmb/bitmark-parser
|
|
|
11
11
|
```js
|
|
12
12
|
import {
|
|
13
13
|
convert,
|
|
14
|
-
|
|
14
|
+
semanticTokens,
|
|
15
15
|
breakscapeText,
|
|
16
16
|
unbreakscapeText,
|
|
17
17
|
info,
|
|
@@ -32,11 +32,18 @@ console.log(parsedJson[0].bitmark);
|
|
|
32
32
|
// Generate bitmark from JSON
|
|
33
33
|
const bitmark = convert(json, { inputFormat: "json", outputFormat: "bitmark" });
|
|
34
34
|
|
|
35
|
-
//
|
|
36
|
-
const
|
|
35
|
+
// Parser-derived highlighting in the LSP semantic-tokens shape
|
|
36
|
+
const highlights = semanticTokens("[.article]\nHello **bold**", {
|
|
37
|
+
tokensLayout: "absolute",
|
|
38
|
+
});
|
|
37
39
|
|
|
38
|
-
//
|
|
39
|
-
const
|
|
40
|
+
// Dump the lexer's token stream (debug tooling; bitmark input only)
|
|
41
|
+
const tokens = JSON.parse(
|
|
42
|
+
convert("[.article]\nHello **bold**", {
|
|
43
|
+
inputFormat: "bitmark",
|
|
44
|
+
outputFormat: "lex",
|
|
45
|
+
}),
|
|
46
|
+
);
|
|
40
47
|
|
|
41
48
|
// Breakscape / unbreakscape text
|
|
42
49
|
const escaped = breakscapeText("[!example]", {
|
|
@@ -60,23 +67,37 @@ console.log(version());
|
|
|
60
67
|
The package ships **three wasm builds of the same engine** and selects one at
|
|
61
68
|
runtime — the API surface never changes:
|
|
62
69
|
|
|
63
|
-
| Feature
|
|
64
|
-
|
|
|
65
|
-
| `full` (Node default)
|
|
66
|
-
| `browser-full` (browser default) | the same conversions and `diff`, without either
|
|
67
|
-
| `bitmark-json`
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
70
|
+
| Feature | Contents | Size (approx.) |
|
|
71
|
+
| -------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------- |
|
|
72
|
+
| `full` (Node default) | everything: rich `info` metadata + built-in translations + the semantic `diff` | ~1025 KB (~395 KB gzip) |
|
|
73
|
+
| `browser-full` (browser default) | the same conversions and `diff`, without either | ~818 KB (~333 KB gzip) |
|
|
74
|
+
| `bitmark-json` | bitmark ↔ JSON only: `convert`/`canonicalize`/`transform` (formats `auto`/`bitmark`/`json`, plus the `text`, `semantic-tokens` and `diagnostics` outputs), the editor services, `info` (JSON only), breakscape, text fragments — no `diff` | ~599 KB (~239 KB gzip) |
|
|
75
|
+
|
|
76
|
+
**The rule.** _Every build includes every feature, except the two
|
|
77
|
+
browser-targeted variants, which each carry an explicit exclusion list._ That
|
|
78
|
+
is the whole story — a new capability is in every build unless it is named
|
|
79
|
+
below:
|
|
80
|
+
|
|
81
|
+
| build | features |
|
|
82
|
+
| -------------------------------------------------- | ----------------------------------------------------------------------------------- |
|
|
83
|
+
| native CLI, daemon, docgen, schemagen, wasm `full` | **all** |
|
|
84
|
+
| wasm `browser-full` | all **except** `info-meta` (bit + tag prose) and `translations` (the baked table) |
|
|
85
|
+
| wasm `bitmark-json` | all **except** the above and `markup`, `mapping-report`, `diff`, `lex`, `info-text` |
|
|
86
|
+
|
|
87
|
+
So, not in `bitmark-json`: the markup formats (`html`, `xml`,
|
|
88
|
+
`xml-niso-iec`, …) as input or output (they throw an `unsupported` error), the
|
|
89
|
+
`lex` output format and the `mappingReport` option (same error), `diff`
|
|
90
|
+
(throws `UnsupportedFeatureError`), and the human-readable rendering of `info` —
|
|
91
|
+
`info({ format: "text" })` throws `UnsupportedFeatureError` there, so pass
|
|
92
|
+
`format: "json"` (the JSON views are the contractual ones and are in every
|
|
93
|
+
variant).
|
|
73
94
|
|
|
74
95
|
`full` and `browser-full` convert **identically** — they differ only in what
|
|
75
|
-
`info` can report: the META fields (tag descriptions, group provenance
|
|
76
|
-
mapping patterns) and the built-in translations table. Bit titles
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
96
|
+
`info` can report: the META fields (bit and tag descriptions, group provenance
|
|
97
|
+
and raw mapping patterns) and the built-in translations table. Bit titles and
|
|
98
|
+
the bit-group / resource-group catalogs are in every variant, in English;
|
|
99
|
+
`register` supplies translations to the lean ones (see _Display names and
|
|
100
|
+
languages_).
|
|
80
101
|
|
|
81
102
|
Why the defaults differ: only the browser pays a download-latency cost for
|
|
82
103
|
those strings, and a Node backend loading wasm from disk has the same
|
|
@@ -89,15 +110,15 @@ browser, select `browser-full` explicitly.**
|
|
|
89
110
|
> **Migrating from ≤ 6.10:** `full` used to mean what `browser-full` means now.
|
|
90
111
|
> Browser code that explicitly selected `"full"` should move to
|
|
91
112
|
> `"browser-full"` to keep its download size close to what it was; leave it on
|
|
92
|
-
> `"full"` to gain the metadata and built-in translations for ~
|
|
113
|
+
> `"full"` to gain the metadata and built-in translations for ~62 KB gzip.
|
|
93
114
|
> Nothing else changes — the capabilities that were in `full` are in both.
|
|
94
115
|
|
|
95
116
|
```js
|
|
96
117
|
import { init, convert, variant } from "@gmb/bitmark-parser";
|
|
97
118
|
|
|
98
119
|
await init({ feature: "bitmark-json" }); // load the minimal wasm
|
|
99
|
-
convert("[.article]\nHello");
|
|
100
|
-
variant();
|
|
120
|
+
convert("[.article]\nHello"); // works
|
|
121
|
+
variant(); // "bitmark-json"
|
|
101
122
|
|
|
102
123
|
// Upgrade in place: the interface stays valid while the full wasm loads —
|
|
103
124
|
// calls keep being served by the active module, then swap atomically.
|
|
@@ -114,8 +135,10 @@ Notes:
|
|
|
114
135
|
the selected variant's `.wasm` is fetched (all three small JS glue modules
|
|
115
136
|
are in the bundle). Overlapping `init` calls: same-feature calls coalesce; a
|
|
116
137
|
different-feature call supersedes an unfinished one (last call wins); every
|
|
117
|
-
promise resolves once the finally-active module is live.
|
|
118
|
-
`initSync(bytes, { feature })`
|
|
138
|
+
promise resolves once the finally-active module is live. `init({})` is the
|
|
139
|
+
same request as `init()`. In the browser, `initSync(bytes, { feature })`
|
|
140
|
+
instantiates from bytes you supply, and takes part in the same last-call-wins
|
|
141
|
+
order: an `init` still loading when `initSync` runs never swaps in afterwards.
|
|
119
142
|
- **Supplying your own `.wasm`**: glue and module are a pair, so name the
|
|
120
143
|
feature the bytes belong to — `init({ feature, module_or_path })`. The
|
|
121
144
|
back-compat positional form `init(module_or_path)` stays on `full`, since
|
|
@@ -130,14 +153,56 @@ Notes:
|
|
|
130
153
|
|
|
131
154
|
### API
|
|
132
155
|
|
|
156
|
+
#### Errors
|
|
157
|
+
|
|
158
|
+
**Errors are thrown, never returned.** Every fallible WASM export raises a
|
|
159
|
+
JavaScript `Error` whose `message` has one shape:
|
|
160
|
+
|
|
161
|
+
```text
|
|
162
|
+
<kind> at <path>: <message>
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
A returned string is therefore always a result — there is no prefix or
|
|
166
|
+
sentinel to test for, and every call site is a `try`/`catch`.
|
|
167
|
+
|
|
168
|
+
`<kind>` is a **stable kebab-case identifier** — the part to match on.
|
|
169
|
+
`<path>` locates the failure (a JSON pointer-ish path, `offset N` for a JSON
|
|
170
|
+
syntax error, or empty) and `<message>` is human-readable prose that may
|
|
171
|
+
change in any release. The kinds:
|
|
172
|
+
|
|
173
|
+
| kind | meaning |
|
|
174
|
+
| -------------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
|
|
175
|
+
| `invalid-json` | the input string is not well-formed JSON |
|
|
176
|
+
| `compose` | the reverse composer rejected the JSON (shape / field / type) |
|
|
177
|
+
| `unsupported` | the requested conversion is not available (an output-only format used as input, or a format compiled out of this variant) |
|
|
178
|
+
| `patch-parse-error` | a patch or patch document is malformed |
|
|
179
|
+
| `patch-index-out-of-range` | a patch path addressed an array element that does not exist |
|
|
180
|
+
| `patch-type-mismatch` | a patch value's JSON type is incompatible with its target |
|
|
181
|
+
| `patch-conflict` | a patch document's entries contradict each other (a bit removed or moved twice, or an anchor that is itself removed or moved) |
|
|
182
|
+
|
|
183
|
+
Every TypeScript wrapper propagates that `Error` unchanged — `convert`,
|
|
184
|
+
`transform`, `canonicalize`, `bitmarkToObjects`, `countBits`, `diff`, … — as
|
|
185
|
+
does the parallel worker pool, which rejects with the first failing bit's
|
|
186
|
+
error. `register` re-wraps it as a `RegisterError`; a capability the loaded
|
|
187
|
+
variant compiled out raises `UnsupportedFeatureError` before any wasm call.
|
|
188
|
+
|
|
133
189
|
#### convert(input: string, options?: ConvertOptions): string
|
|
134
190
|
|
|
135
191
|
Convert between bitmark, JSON, and the mapped markup formats (html/xml/…).
|
|
136
192
|
|
|
193
|
+
A JSON input may be an array or a single value, each entry a `{ bit: {…} }`
|
|
194
|
+
envelope (the forward output) or a bare bit object. An entry that is not a
|
|
195
|
+
bit produces no bitmark: it is skipped and reported as an `unknown-bit-type`
|
|
196
|
+
issue naming its JSON path (`$[2] is not a bit: no "type" — skipped`) on the
|
|
197
|
+
warnings channel of the CLI (`--warnings`) and the daemon; the bits beside it
|
|
198
|
+
still convert, and a document in which nothing is a bit converts to empty
|
|
199
|
+
output. This `convert` returns a bare string and so cannot show the skip.
|
|
200
|
+
Only malformed JSON *text* throws (`invalid-json`).
|
|
201
|
+
|
|
137
202
|
Options:
|
|
138
203
|
|
|
139
204
|
- `inputFormat` — `"auto"` (default), `"bitmark"`, `"json"`, or a mapping id (`"html"`, `"xml"`, …)
|
|
140
|
-
- `outputFormat` — `"auto"` (default: the opposite direction), `"text"` (lossy plain-text extraction; output-only), or as above
|
|
205
|
+
- `outputFormat` — `"auto"` (default: the opposite direction), `"text"` (lossy plain-text extraction; output-only), `"semantic-tokens"` / `"diagnostics"` / `"lex"` (output-only, bitmark input only — see their sections), or as above
|
|
141
206
|
- `mode` — `"optimized"` (default) or `"full"`
|
|
142
207
|
- `warnings` — include validation warnings (default: `false`)
|
|
143
208
|
- `plainText` — output text as plain text (default: `false`)
|
|
@@ -153,7 +218,7 @@ Options:
|
|
|
153
218
|
the bit is `_`-prefixed. A bare tag the bit does not declare (`[!…]`,
|
|
154
219
|
`[?…]`, `[#…]`, … on a bit whose config has no such tag) is an unknown
|
|
155
220
|
too, under the key a declaring bit would use (`[!x]` → `"instruction":
|
|
156
|
-
|
|
221
|
+
["x"]`); a rejected resource attachment is not — it is still emitted, with
|
|
157
222
|
a warning. Unknown properties are never converted back to bitmark, and
|
|
158
223
|
each occurrence warns whether or not it is included.
|
|
159
224
|
Also available on `bitmarkToObjects` and as the CLI's
|
|
@@ -175,6 +240,13 @@ Options:
|
|
|
175
240
|
> format, set `inputFormat` explicitly. The same applies to `canonicalize`
|
|
176
241
|
> and `transform`, whose `inputFormat` also defaults to `"auto"`.
|
|
177
242
|
|
|
243
|
+
> **Input limits:** JSON objects/arrays and markup elements may nest at most
|
|
244
|
+
> **256** levels deep. Deeper JSON is rejected like any other malformed JSON
|
|
245
|
+
> (`"error: … nesting depth exceeds 256"`); a deeper markup open tag is kept
|
|
246
|
+
> as literal text. The bound keeps a crafted document from exhausting the
|
|
247
|
+
> WASM stack — an instance that rejects a too-deep input stays usable for the
|
|
248
|
+
> next call.
|
|
249
|
+
|
|
178
250
|
`convert` is the single string-based conversion surface — parsing is
|
|
179
251
|
`{ inputFormat: "bitmark", outputFormat: "json" }`, generating is
|
|
180
252
|
`{ inputFormat: "json", outputFormat: "bitmark" }`. For typed objects use
|
|
@@ -183,8 +255,16 @@ Options:
|
|
|
183
255
|
For bitmark input, output is a JSON array of entries with:
|
|
184
256
|
|
|
185
257
|
- `bit` — serialized bit payload
|
|
186
|
-
- `parser` — parser metadata, `errors` / `warnings`, and `infos` (informational notices for recoveries that are not necessarily wrong
|
|
187
|
-
- `bitmark` — source bitmark text
|
|
258
|
+
- `parser` — parser metadata, `errors` / `warnings`, and `infos` (informational notices for recoveries that are not necessarily wrong: text kept as text because it only looks like an unclosed tag, `unclosed-tag`, or a formatting mark with no partner, `unclosed-formatting` — in bitmark malformed markup is text, not an error). Each issue carries a stable machine-readable `code` (`"unknown-property"`, `"missing-required-tag"`, …) beside its human-readable `message`, plus `text` and `location`. **Branch on `code`** — a shipped code never changes meaning, whereas the message text is not contractual and may be reworded in any release
|
|
259
|
+
- `bitmark` — source bitmark text (a convenience echo; every bit is fully described by `bit` alone)
|
|
260
|
+
|
|
261
|
+
A region the parser could not read — an unknown bit type, or non-blank text
|
|
262
|
+
that never opened a bit — comes back as a bit of type `_error` beside its
|
|
263
|
+
`parser.errors` entry. It carries `originalType` (the header exactly as
|
|
264
|
+
written between the level dots and the `]`, including a leading `|` for a
|
|
265
|
+
commented-out bit and any `:format` / `&resource` suffix; absent when there
|
|
266
|
+
was no header) and `body` (the raw text after it, as a plain string), so
|
|
267
|
+
generating bitmark from it reproduces the original region.
|
|
188
268
|
|
|
189
269
|
#### canonicalize(input: string, options?: CanonicalizeOptions): string
|
|
190
270
|
|
|
@@ -203,7 +283,13 @@ Apply a patch document to a document, bit by bit, and re-emit it.
|
|
|
203
283
|
before the document reshapes anything; a hook's patches apply after the
|
|
204
284
|
document's
|
|
205
285
|
- `inputFormat`, `outputFormat`, `mode`, `pretty`, `indent`,
|
|
206
|
-
`spacesAroundValues` — as for `convert`
|
|
286
|
+
`spacesAroundValues` — as for `convert`. `outputFormat` (default `"json"`)
|
|
287
|
+
takes every per-bit format: `"json"`, `"bitmark"`, `"text"` or a mapping id
|
|
288
|
+
such as `"html"` (per-bit text outputs are joined with newlines). The
|
|
289
|
+
whole-document outputs `"lex"`, `"semantic-tokens"` and `"diagnostics"` are
|
|
290
|
+
refused with an `unsupported` error in every driver — use `convert`. An
|
|
291
|
+
explicit `inputFormat` is honoured as given, patch or hooks or not; it is
|
|
292
|
+
never re-detected from the content
|
|
207
293
|
|
|
208
294
|
A patch document addresses bits by their position (or id) in the input
|
|
209
295
|
**before** any entry is applied, so it reads like a diff:
|
|
@@ -212,9 +298,15 @@ A patch document addresses bits by their position (or id) in the input
|
|
|
212
298
|
{
|
|
213
299
|
"version": 1,
|
|
214
300
|
"entries": [
|
|
215
|
-
{
|
|
301
|
+
{
|
|
302
|
+
"select": { "index": 3, "id": "mc-014" },
|
|
303
|
+
"patch": [{ "path": "title", "op": "set", "value": "New" }]
|
|
304
|
+
},
|
|
216
305
|
{ "select": { "index": 9 }, "remove": true },
|
|
217
|
-
{
|
|
306
|
+
{
|
|
307
|
+
"insert": { "after": 4 },
|
|
308
|
+
"bit": { "type": "article", "body": "Inserted" }
|
|
309
|
+
},
|
|
218
310
|
{ "select": { "index": 7 }, "move": { "after": 12 } },
|
|
219
311
|
{ "select": "all", "patch": [{ "lang": "de" }] }
|
|
220
312
|
]
|
|
@@ -222,8 +314,10 @@ A patch document addresses bits by their position (or id) in the input
|
|
|
222
314
|
```
|
|
223
315
|
|
|
224
316
|
`after` names an anchor — an input bit that is neither removed nor moved,
|
|
225
|
-
or `-1` for the document start.
|
|
226
|
-
|
|
317
|
+
or `-1` for the document start. A bit may be the target of at most one
|
|
318
|
+
`remove` or `move` entry: moving a bit twice, removing it twice, or removing
|
|
319
|
+
and moving it are all rejected, naming the second of the conflicting entries.
|
|
320
|
+
`diff` produces such a document from two versions of a file (below).
|
|
227
321
|
|
|
228
322
|
#### diff(a: string, b: string, options?: DiffOptions): string
|
|
229
323
|
|
|
@@ -279,16 +373,195 @@ issue codes gained or lost, and the rendered `hunks`.
|
|
|
279
373
|
|
|
280
374
|
#### countBits(input: string): number / splitBits(input: string): BitSlice[]
|
|
281
375
|
|
|
282
|
-
Count or split the bits of a bitmark/JSON input without full parsing.
|
|
376
|
+
Count or split the bits of a bitmark/JSON input without full parsing. A JSON
|
|
377
|
+
input may be an array or a single value, each entry a `{ bit: {…} }` envelope
|
|
378
|
+
or a bare bit object (the four shapes `convert` accepts); `splitJsonBits`
|
|
379
|
+
always returns envelopes. An entry that is not a bit is not counted and not
|
|
380
|
+
returned — these calls have no warnings channel; `convert` reports the skip.
|
|
381
|
+
|
|
382
|
+
#### semanticTokens(input: string, options?: SemanticTokensOptions): SemanticTokens
|
|
383
|
+
|
|
384
|
+
Parser-derived highlighting of bitmark source in the LSP semantic-tokens
|
|
385
|
+
shape — the typed form of
|
|
386
|
+
`convert(input, { inputFormat: "bitmark", outputFormat: "semantic-tokens" })`.
|
|
387
|
+
The parser runs (never the validator; diagnostics stay on their own channel)
|
|
388
|
+
and the AST is walked loss-lessly, so every character of the source is
|
|
389
|
+
covered exactly once and the highlighting agrees with the JSON the same
|
|
390
|
+
input produces: an unpaired `**` is text, an abandoned `[!tag` is text and
|
|
391
|
+
the `====` / `--` it would have swallowed are dividers, the body of a bit
|
|
392
|
+
whose body format is not bitmark (`[.code]`, `==== text ====`) is plain
|
|
393
|
+
text. Bitmark input only.
|
|
394
|
+
|
|
395
|
+
Options (`SemanticTokensOptions`, also `ConvertOptions` keys):
|
|
396
|
+
|
|
397
|
+
- `positionEncoding` — `"utf-16"` (default; the LSP default, what browser
|
|
398
|
+
editors index by) or `"utf-8"` (bytes). Echoed in the response.
|
|
399
|
+
- `tokensLayout` — `"lsp"` (default): `data`, the LSP relative encoding,
|
|
400
|
+
five integers per token (`deltaLine, deltaStart, length, tokenType,
|
|
401
|
+
tokenModifiers`; the type is a legend index, the modifiers a bitset); or
|
|
402
|
+
`"absolute"`: `tokens`, one `{ line, start, length, type, modifiers }`
|
|
403
|
+
per token, 0-based, by name. Both carry `legend` and `positionEncoding`.
|
|
404
|
+
|
|
405
|
+
Lines break at `\n`, `\r\n` and `\r`; no token contains a line terminator.
|
|
406
|
+
The legend orders are stable (a change is semver-major; new entries may be
|
|
407
|
+
appended). Token types name sigil classes and syntax positions, never a bit
|
|
408
|
+
or tag name; each has a recommended standard super type so an editor's
|
|
409
|
+
default theme colours it — paste the table into a VS Code extension's
|
|
410
|
+
`semanticTokenTypes`:
|
|
411
|
+
|
|
412
|
+
| type | covers | super type |
|
|
413
|
+
| ---------------------------------------------------------------------------------- | -------------------------------------------------------- | ---------------------------------- |
|
|
414
|
+
| `frontmatter` | the region before the first bit | `comment` |
|
|
415
|
+
| `bitSigil` | `[.`, level dots, `\|`, `:`, `&`, `]` of a bit header | `keyword` |
|
|
416
|
+
| `bitType` / `bitFormat` / `bitResourceType` | the header's three fields | `type` / `modifier` / `type` |
|
|
417
|
+
| `tagSigil` | every tag's own syntax (`[@`, `[!`, `[##`, `:`, `]`, …) | `keyword` |
|
|
418
|
+
| `propertyKey` / `resourceType` | a property tag's key / a resource tag's type | `property` / `type` |
|
|
419
|
+
| `tagText` | plain text inside any tag value | `string` |
|
|
420
|
+
| `cardDivider` / `sideDivider` / `variantDivider` / `footerDivider` / `textDivider` | `====`, `--`, `++`, `==== footer ====`, `==== text ====` | `operator` |
|
|
421
|
+
| `paragraphBreak` | a `\|` paragraph break | `operator` |
|
|
422
|
+
| `headingSigil` / `heading` | `# ` / the heading's plain text | `keyword` / `string` |
|
|
423
|
+
| `listMarker` | `• `, `•1 `, `•+ `, the indent | `keyword` |
|
|
424
|
+
| `codeSigil` / `codeLanguage` / `codeBody` | `\|code:` / the language / the body | `keyword` / `type` / `string` |
|
|
425
|
+
| `imageSigil` / `imageSrc` | `\|image:` / the source | `keyword` / `string` |
|
|
426
|
+
| `markSigil` | a mark's open and close markers | `operator` |
|
|
427
|
+
| `bold` / `italic` / `highlight` / `light` / `inline` | the marked text (in body text and `bitmark++` values; a string value is one literal `tagText`) | `string` |
|
|
428
|
+
| `attrSigil` / `attrKey` / `attrValue` | an attr chain's `\|` and `:` / key / value | `operator` / `property` / `string` |
|
|
429
|
+
| `url` | a bare URL the parser auto-linked | `string` |
|
|
430
|
+
| `text` | plain body / card / footer text, and every top-level gap | (none) |
|
|
431
|
+
| `plainText` | a plain region (after `==== text ====`, or a raw body) | `string` |
|
|
432
|
+
|
|
433
|
+
Modifiers, in bit order: the tag classes `property`, `resource`, `title`,
|
|
434
|
+
`reference`, `anchor`, `item`, `instruction`, `hint`, `true`, `false`,
|
|
435
|
+
`gap`, `mark`, `solution` (every token of a tag of that class, sigils
|
|
436
|
+
included — exactly one per tag); `comment` (every token of a commented-out
|
|
437
|
+
bit); `unclosed` (tokens inside a parser unclosed-mark / unclosed-tag
|
|
438
|
+
notice); `bold`, `italic`, `highlight`, `light` (the text of an inline mark
|
|
439
|
+
`==…==\|…\|` whose chain carries that bare key — style it as you style the
|
|
440
|
+
`bold` / `italic` / `highlight` / `light` token types). The full contract is
|
|
441
|
+
`.zen/specs/API-SEM-semantic-tokens.tsp`.
|
|
442
|
+
|
|
443
|
+
#### diagnostics(input: string, options?: DiagnosticsOptions): Diagnostics
|
|
444
|
+
|
|
445
|
+
Every issue the JSON `parser` envelope would carry for bitmark source, as
|
|
446
|
+
LSP 3.17 `Diagnostic`s with editor positions — the typed form of
|
|
447
|
+
`convert(input, { inputFormat: "bitmark", outputFormat: "diagnostics" })`.
|
|
448
|
+
Parse and validate, no serialization: cheap enough for every keystroke. The
|
|
449
|
+
validator owns every issue, so this is exactly the envelope's set — the same
|
|
450
|
+
`code`s, the same spans, the same order (bit by bit) — with the envelope's
|
|
451
|
+
buckets as LSP severities: `errors` → 1 (`DiagnosticSeverity.Error`),
|
|
452
|
+
`warnings` → 2, `infos` → 3. Bitmark input only.
|
|
283
453
|
|
|
284
|
-
|
|
454
|
+
```js
|
|
455
|
+
import { diagnostics, DiagnosticSeverity } from "@gmb/bitmark-parser";
|
|
456
|
+
|
|
457
|
+
const { positionEncoding, diagnostics: list } = diagnostics(source);
|
|
458
|
+
// list[i] = {
|
|
459
|
+
// range: { start: { line, character }, end: { line, character } }, // 0-based, UTF-16 units
|
|
460
|
+
// severity: 2, // DiagnosticSeverity.Warning
|
|
461
|
+
// code: "unknown-property", // the contract — branch on this
|
|
462
|
+
// source: "bitmark",
|
|
463
|
+
// message: "[@nope] is an unknown property …", // prose, may change
|
|
464
|
+
// data: { bit: 0 }, // the bit's index in the JSON array
|
|
465
|
+
// }
|
|
466
|
+
```
|
|
467
|
+
|
|
468
|
+
Option: `positionEncoding` — `"utf-16"` (default) or `"utf-8"`, as for
|
|
469
|
+
`semanticTokens`; echoed in the response. The types (`Diagnostic`, `Range`,
|
|
470
|
+
`Position`, `DiagnosticSeverity`) are written field for field as the
|
|
471
|
+
protocol defines them, so a `Diagnostic` assigns to Monaco's marker input
|
|
472
|
+
and to `vscode-languageserver-types` without a cast, and the package keeps
|
|
473
|
+
its zero runtime dependencies. In every wasm variant. The contract is
|
|
474
|
+
`.zen/specs/API-EDT-editor-services.tsp`.
|
|
475
|
+
|
|
476
|
+
#### complete(input, position, options?) / resolve(input, position, item, options?) / hover(input, position, options?)
|
|
477
|
+
|
|
478
|
+
The editor position queries, in LSP 3.17 shapes (PLAN-196, PLAN-202). Positions are
|
|
479
|
+
0-based `{ line, character }` in `positionEncoding` units — `"utf-16"` by
|
|
480
|
+
default, as for `semanticTokens`.
|
|
481
|
+
|
|
482
|
+
`complete` returns an LSP `CompletionList` of everything valid at the cursor:
|
|
483
|
+
|
|
484
|
+
| where the cursor is | what you get |
|
|
485
|
+
| --- | --- |
|
|
486
|
+
| a bit header's type (`[.art`) | every bit type (`kind: Class`), with its title |
|
|
487
|
+
| a header's `:format` / `&resource` | the body formats / the resource types this bit allows |
|
|
488
|
+
| a tag, or body / card / footer text | the tags valid in THAT scope — the bit body, a card variant, a chain — minus those already at their maximum, the ones the scope still requires sorted first (the first `preselect`ed) |
|
|
489
|
+
| a tag's value | an `enum` tag's vocabulary, a `boolean` tag's `true` / `false` |
|
|
490
|
+
| after a tag's `]` | that tag's chain children |
|
|
491
|
+
| an inline attribute chain (`==x==\|bo`) | the attribute keys, and the closed value set of one that has it |
|
|
492
|
+
| a line holding a prefix of one (empty, `=`, `==== f`) | the structural lines the bit's card set allows there (`====`, `--`, `++`), `==== footer ====`, `==== text ====` — each documented |
|
|
493
|
+
| bitmark text (body text, a `bitmark++` tag's value), after none, `=` or `==` | the inline mark `==…==\|…\|` as a snippet (`insertTextFormat: 2`); never in a string value or a plain region |
|
|
494
|
+
|
|
495
|
+
Items carry `kind`, `detail`, `tags: [1]` when deprecated, `sortText`,
|
|
496
|
+
`insertText`, and — in LSP's own `data` slot — the `info` record behind
|
|
497
|
+
them, so you can build UI from fields rather than prose. `isIncomplete` is
|
|
498
|
+
always `false`: every candidate for the context comes back and filtering by
|
|
499
|
+
the typed prefix is the editor's job. A half-typed `[@wi`, `[.art` or
|
|
500
|
+
`==x==\|bo` resolves as if it were closed. A header resource attachment
|
|
501
|
+
(`[.flashcard&image]`) is offered at the bit's level until it is there; a
|
|
502
|
+
tag whose label is only its sigil (`%`, `!`) leads its `detail` with its
|
|
503
|
+
name (`item`, `instruction`). The result adds one field beyond LSP,
|
|
504
|
+
`context`, saying where the query resolved.
|
|
505
|
+
|
|
506
|
+
Pass `triggerCharacter` in the options when a character opened the query
|
|
507
|
+
(LSP `CompletionContext.triggerCharacter`; Monaco and VS Code hand it to the
|
|
508
|
+
provider). A trigger that opens nothing where the cursor is — a `.` or a
|
|
509
|
+
`-` typed in prose — answers an empty list; `[` opens a tag anywhere, and
|
|
510
|
+
`=` / `-` / `+` narrow the list to the structural lines they begin. An
|
|
511
|
+
explicit invocation (Ctrl+Space) is never narrowed. Editors that open the
|
|
512
|
+
list on every letter (Monaco's `quickSuggestions`) should turn that off for
|
|
513
|
+
bitmark: it is prose with markup in it.
|
|
514
|
+
|
|
515
|
+
Items carry NO `documentation` (PLAN-202): a header's list is every bit type,
|
|
516
|
+
and its prose would be most of the bytes of every list. `resolve` is LSP's
|
|
517
|
+
`completionItem/resolve` — pass the same `input` and `position` you gave
|
|
518
|
+
`complete` plus one item as it was listed, and get that item back with its
|
|
519
|
+
Markdown `documentation` (a bit's title and description; a tag's format,
|
|
520
|
+
count, default, JSON key, chain and description in this scope; an inline
|
|
521
|
+
attribute's shape). Wire it to Monaco's `resolveCompletionItem` or an LSP
|
|
522
|
+
server's resolve handler, so only the item about to be shown is rendered.
|
|
523
|
+
When the list at that position no longer holds the item, the item comes
|
|
524
|
+
back as given.
|
|
525
|
+
|
|
526
|
+
`hover` returns an LSP `Hover` — Markdown `contents` plus the record in
|
|
527
|
+
`data` — for the bit type, a tag AS RESOLVED IN ITS SCOPE (the same record
|
|
528
|
+
`info({ infoType: "bit", full: true })` shows for it there), a tag value, a
|
|
529
|
+
header resource type, or an inline attribute key or value. It is `null` on
|
|
530
|
+
body text, dividers and whitespace.
|
|
531
|
+
|
|
532
|
+
```js
|
|
533
|
+
import { complete, resolve, hover, CompletionItemKind } from "@gmb/bitmark-parser";
|
|
534
|
+
|
|
535
|
+
const list = complete("[.article]\n[@", { line: 1, character: 2 });
|
|
536
|
+
list.context; // { bit: "article", scope: "bit" }
|
|
537
|
+
list.items.filter((i) => i.kind === CompletionItemKind.Property).map((i) => i.label);
|
|
285
538
|
|
|
286
|
-
|
|
539
|
+
const id = list.items.find((i) => i.label === "@id");
|
|
540
|
+
resolve("[.article]\n[@", { line: 1, character: 2 }, id).documentation.value; // Markdown for @id here
|
|
287
541
|
|
|
288
|
-
|
|
542
|
+
const h = hover("[.article]\n[@id:1]", { line: 1, character: 2 });
|
|
543
|
+
h.contents.value; // Markdown: format, count, default, JSON key, description
|
|
544
|
+
h.data.tag; // the info record for @id in this bit
|
|
545
|
+
```
|
|
289
546
|
|
|
290
|
-
|
|
291
|
-
|
|
547
|
+
Both need a variant built with `editor` (all three are) — otherwise they
|
|
548
|
+
throw `UnsupportedFeatureError`. All three read bitmark SOURCE, so a JSON
|
|
549
|
+
document is refused; an empty buffer, or one that is still only frontmatter,
|
|
550
|
+
is source being typed and is answered normally (an empty one's `complete`
|
|
551
|
+
offers the bit types). The CLI has the same three services under one
|
|
552
|
+
command: `bitmark editor diagnostics|complete|resolve|hover`. The contract is
|
|
553
|
+
`.zen/specs/API-EDT-editor-services.tsp`.
|
|
554
|
+
|
|
555
|
+
#### The `lex` output format
|
|
556
|
+
|
|
557
|
+
`convert(input, { inputFormat: "bitmark", outputFormat: "lex" })` returns the
|
|
558
|
+
lexer's token stream as a JSON array: one `{ kind, span, text }` per token
|
|
559
|
+
(the `LexToken` type), `span` in bytes. Debug tooling — the token kinds are the
|
|
560
|
+
engine's own names, not a stable vocabulary, and the parser is never run
|
|
561
|
+
(pairing, abandoned-tag repair and raw-body handling all happen after lexing),
|
|
562
|
+
so a highlighter should use `semantic-tokens` instead. Bitmark input only; a
|
|
563
|
+
JSON or markup document has no source text to lex. The former `lex()` export
|
|
564
|
+
and `bitmark lex` command were removed in 7.0 (see the migration guide).
|
|
292
565
|
|
|
293
566
|
#### breakscapeText(input: string, options?: BreakscapeOptions): string
|
|
294
567
|
|
|
@@ -325,13 +598,13 @@ version and (separately) any migration target.
|
|
|
325
598
|
- `full` — complete detail for `"bit"`/`"all"` (default: `false` = compact view)
|
|
326
599
|
- `language` — BCP-47 tag(s) for display names: one (`"de"`), several
|
|
327
600
|
(`["de", "fr"]` or `"de,fr"`), or `"all"` — exported as `ALL_LANGUAGES`
|
|
328
|
-
(default: English). More than one adds a `titles` map. See
|
|
329
|
-
and
|
|
601
|
+
(default: English). More than one adds a `titles` map. See _Display names
|
|
602
|
+
and languages_ below
|
|
330
603
|
|
|
331
604
|
`"bit-groups"` / `"resource-groups"` return the search/filter catalogs —
|
|
332
605
|
each group's key, translated title, description, optional aliases and
|
|
333
606
|
`subgroupOf`, and its members. They are available in EVERY build: the group
|
|
334
|
-
catalog and per-bit titles
|
|
607
|
+
catalog and per-bit titles are "descriptive" data, which the
|
|
335
608
|
lean wasm variants carry too. Deprecated members are excluded by default and
|
|
336
609
|
MARKED when included with `includeDeprecated: true`; a search index matching
|
|
337
610
|
already-published content needs them, because a migrating bit is re-emitted
|
|
@@ -341,13 +614,13 @@ To derive quiz categories, intersect a bit's `bitGroups` with the groups
|
|
|
341
614
|
carrying `subgroupOf: "quizzes"` — `subgroupOf` is metadata and never implies
|
|
342
615
|
membership, so every member of a subgroup also declares the parent.
|
|
343
616
|
|
|
344
|
-
Only the META fields — TAG descriptions, group-inheritance provenance
|
|
345
|
-
mapping patterns (the `info-meta` cargo feature) — depend on the build.
|
|
346
|
-
are reported by the native CLI and the wasm `full` variant; `browser-full`
|
|
347
|
-
`bitmark-json` omit them to stay small
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
variant.
|
|
617
|
+
Only the META fields — BIT and TAG descriptions, group-inheritance provenance
|
|
618
|
+
and raw mapping patterns (the `info-meta` cargo feature) — depend on the build.
|
|
619
|
+
They are reported by the native CLI and the wasm `full` variant; `browser-full`
|
|
620
|
+
and `bitmark-json` omit them to stay small (the bit descriptions alone are
|
|
621
|
+
14 KB gzip of the download). Those are absent structurally: the key is
|
|
622
|
+
missing, never `null` or empty. Everything else `info` returns — including bit
|
|
623
|
+
titles and the group catalogs — is identical in every variant.
|
|
351
624
|
|
|
352
625
|
#### Display names and languages
|
|
353
626
|
|
|
@@ -390,7 +663,12 @@ which categories it is in":
|
|
|
390
663
|
|
|
391
664
|
```js
|
|
392
665
|
const groups = JSON.parse(
|
|
393
|
-
info({
|
|
666
|
+
info({
|
|
667
|
+
infoType: "bit-groups",
|
|
668
|
+
format: "json",
|
|
669
|
+
includeDeprecated: true,
|
|
670
|
+
language: "all",
|
|
671
|
+
}),
|
|
394
672
|
);
|
|
395
673
|
groups[0].bitTypes[0]; // { name: "assignment", title: "Assignment", titles: {…} }
|
|
396
674
|
```
|
|
@@ -508,7 +786,7 @@ const patch = patchEntry("id", "append", "1234");
|
|
|
508
786
|
|
|
509
787
|
See `examples/` for runnable scripts covering the whole API (executed by
|
|
510
788
|
`npm test`, so they stay correct). Generated API documentation (TypeDoc) is
|
|
511
|
-
published at <https://getmorebrain.github.io/bitmark-parser
|
|
789
|
+
published at <https://getmorebrain.github.io/bitmark-parser/docs/api/>
|
|
512
790
|
with each release (the site root hosts the bitmark language docs), or build
|
|
513
791
|
it locally with `npm run docs` (→ `docs/api/`) and preview it with
|
|
514
792
|
`npm run docs:serve` (http://localhost:8080, `--port` to change).
|
|
@@ -536,6 +814,28 @@ Note: deprecated bit types with a configured migration target (the
|
|
|
536
814
|
`isCollapsible: true` — consumers keyed on the old `type` strings see the
|
|
537
815
|
base type instead. `collapsible` itself is unaffected.
|
|
538
816
|
|
|
817
|
+
### Property number values
|
|
818
|
+
|
|
819
|
+
A tag whose configured format is `number` accepts what JavaScript's
|
|
820
|
+
`Number(value)` accepts, minus the values that are not finite:
|
|
821
|
+
|
|
822
|
+
- surrounding whitespace, a leading `+`, leading zeros (`01`), a trailing
|
|
823
|
+
point (`1.`) and a leading point (`.5`);
|
|
824
|
+
- an exponent in either case (`1e2`, `1E2`);
|
|
825
|
+
- hexadecimal, binary and octal literals (`0x1F`, `0b101`, `0o701`) — prefix
|
|
826
|
+
case-insensitive, with no sign, fraction or exponent.
|
|
827
|
+
|
|
828
|
+
`NaN`, `Infinity`, an exponent beyond ±4000, and a radix literal past 64 bits
|
|
829
|
+
are **rejected**: the tag is dropped from the output and a
|
|
830
|
+
`property-format-mismatch` warning names it, exactly like any other value that
|
|
831
|
+
does not fit its format.
|
|
832
|
+
|
|
833
|
+
Accepted values print in JavaScript's `JSON.stringify` form — `0.5`, `1.5`,
|
|
834
|
+
`100`, `1e+21`, `1e-7` — so `[@width: 1.50]` emits `1.5` and `[@width:0x10]`
|
|
835
|
+
emits `16`. Values with more than 17 significant digits, including integers at
|
|
836
|
+
or above 2^63, keep the digits you wrote rather than being rounded through a
|
|
837
|
+
floating-point double.
|
|
838
|
+
|
|
539
839
|
## Legacy API (bpg-compatible)
|
|
540
840
|
|
|
541
841
|
A compatibility facade for consumers migrating from
|
|
@@ -643,8 +943,12 @@ bitmark diff old.bitmark new.bitmark --output-format patch > d.patch
|
|
|
643
943
|
bitmark transform old.bitmark --patch d.patch --output-format bitmark # reproduces new
|
|
644
944
|
# --context N, --locate, --similarity F, --bbox-tolerance N, --color auto|always|never
|
|
645
945
|
|
|
646
|
-
#
|
|
647
|
-
bitmark
|
|
946
|
+
# Parser-derived highlighting: LSP semantic tokens (bitmark input only)
|
|
947
|
+
bitmark convert input.bitmark --input-format bitmark --output-format semantic-tokens
|
|
948
|
+
bitmark convert input.bitmark --output-format semantic-tokens --tokens-layout absolute --position-encoding utf-8 --pretty
|
|
949
|
+
|
|
950
|
+
# Dump the lexer token stream as JSON (debug tooling; bitmark input only)
|
|
951
|
+
bitmark convert input.bitmark --input-format bitmark --output-format lex --pretty
|
|
648
952
|
|
|
649
953
|
# Breakscape / unbreakscape text
|
|
650
954
|
bitmark breakscape input.txt
|
|
@@ -687,13 +991,13 @@ The package includes pre-built browser bundles with the WASM module.
|
|
|
687
991
|
The CDN bundle carries all three variants' JS glue and fetches only the
|
|
688
992
|
selected variant's `.wasm` (from `dist/browser/wasm/`). A bare `init()` loads
|
|
689
993
|
`browser-full`; `init({ feature: "bitmark-json" })` fetches the smallest
|
|
690
|
-
build, and a later `init({ feature: "full" })` upgrades in place (see
|
|
691
|
-
|
|
994
|
+
build, and a later `init({ feature: "full" })` upgrades in place (see _WASM
|
|
995
|
+
variants_).
|
|
692
996
|
|
|
693
997
|
### Bundler (webpack / vite)
|
|
694
998
|
|
|
695
999
|
```js
|
|
696
|
-
import init, { convert
|
|
1000
|
+
import init, { convert } from "@gmb/bitmark-parser/browser";
|
|
697
1001
|
|
|
698
1002
|
await init();
|
|
699
1003
|
const json = convert("[.article]\nHello", { inputFormat: "bitmark" });
|