@gmb/bitmark-parser 7.5.0 → 7.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -1
- package/config/bitmark.json +522 -170
- package/dist/browser/bitmark-parser.min.js +6 -6
- package/dist/browser/bitmark-parser.min.js.map +1 -1
- package/dist/browser/cjs/index.cjs +669 -241
- package/dist/browser/cjs/index.cjs.map +1 -1
- package/dist/browser/cjs/index.d.cts +129 -8
- package/dist/browser/esm/index.d.ts +129 -8
- package/dist/browser/esm/index.js +668 -241
- package/dist/browser/esm/index.js.map +1 -1
- package/dist/browser/esm/worker-entry.js +626 -233
- package/dist/browser/esm/worker-entry.js.map +1 -1
- package/dist/browser/wasm/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_json_wasm_bg.wasm +0 -0
- package/dist/browser/wasm/bitmark_wasm_bg.wasm +0 -0
- package/dist/index.cjs +39 -4
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +129 -8
- package/dist/index.d.ts +129 -8
- package/dist/index.js +38 -4
- package/dist/index.js.map +1 -1
- package/dist/legacy.cjs +1 -1
- package/dist/legacy.cjs.map +1 -1
- package/dist/legacy.js +1 -1
- package/dist/legacy.js.map +1 -1
- package/dist/worker-entry.cjs.map +1 -1
- package/package.json +7 -7
- package/schema/bitmark.schema.json +133 -1
- package/translations/translations.json +189 -161
- package/wasm/bitmark_wasm.d.ts +65 -31
- package/wasm/bitmark_wasm.js +288 -122
- package/wasm/bitmark_wasm_bg.wasm +0 -0
- package/wasm/bitmark_wasm_bg.wasm.d.ts +6 -1
- package/wasm/package.json +1 -1
- package/wasm-bitmark-json/bitmark_json_wasm.d.ts +72 -38
- package/wasm-bitmark-json/bitmark_json_wasm.js +301 -135
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm +0 -0
- package/wasm-bitmark-json/bitmark_json_wasm_bg.wasm.d.ts +6 -1
- package/wasm-bitmark-json/package.json +1 -1
- package/wasm-browser-full/bitmark_browser_full_wasm.d.ts +65 -31
- package/wasm-browser-full/bitmark_browser_full_wasm.js +288 -122
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm +0 -0
- package/wasm-browser-full/bitmark_browser_full_wasm_bg.wasm.d.ts +6 -1
- package/wasm-browser-full/package.json +1 -1
package/README.md
CHANGED
|
@@ -286,6 +286,48 @@ often than it may be (`|?a|?b|` — the later value wins), and
|
|
|
286
286
|
`text-tag-value-invalid` when a value does not fit its format but is kept
|
|
287
287
|
(`|symbol:…|width:wide|`).
|
|
288
288
|
|
|
289
|
+
#### convertWithDetails(input: string, options?: ConvertWithDetailsOptions): ConvertResult
|
|
290
|
+
|
|
291
|
+
`convert`, returning `{ output, …extras }` — the same conversion, run once,
|
|
292
|
+
with each extra asked for by a flag in the options. `output` is exactly what
|
|
293
|
+
`convert` returns, and it throws exactly as `convert` does. Future extras are
|
|
294
|
+
one more flag and one more field each.
|
|
295
|
+
|
|
296
|
+
- `bitSpans` — `true` adds `result.bitSpans`, where each top-level bit's text
|
|
297
|
+
landed in the output, for linking a source pane to an output pane bit by
|
|
298
|
+
bit:
|
|
299
|
+
|
|
300
|
+
```ts
|
|
301
|
+
const { output, bitSpans } = convertWithDetails(source, {
|
|
302
|
+
inputFormat: "bitmark",
|
|
303
|
+
outputFormat: "html",
|
|
304
|
+
bitSpans: true,
|
|
305
|
+
});
|
|
306
|
+
// bitSpans = { positionEncoding: "utf-16", spans: [{ index, start, end }, …] }
|
|
307
|
+
for (const { index, start, end } of bitSpans!.spans) {
|
|
308
|
+
console.log(index, output.slice(start, end));
|
|
309
|
+
}
|
|
310
|
+
```
|
|
311
|
+
|
|
312
|
+
- `index` is the bit's position in the **input**: its `splitBits` index for
|
|
313
|
+
bitmark input, its entry index in the top-level array for JSON input. An
|
|
314
|
+
entry that is not a bit gets no span, so the indexes can have gaps.
|
|
315
|
+
- `start` / `end` are **the bit's text**, tight: never the separators
|
|
316
|
+
between bits (`\n`, blank lines, `,`) nor the JSON array's `[` / `]`. They
|
|
317
|
+
are counted in `positionEncoding` — `"utf-16"` by default, so
|
|
318
|
+
`output.slice(start, end)` is the bit — or `"utf-8"` bytes.
|
|
319
|
+
- Spans are reported for the document outputs: `bitmark`, `json` (compact or
|
|
320
|
+
pretty), `text`, and the markup formats (`html`, `xml`, …). In `bitmark`
|
|
321
|
+
output they equal `splitBits(output)`'s `start` / `end`.
|
|
322
|
+
- A bit that writes nothing (an empty or `_error` bit in `text`) gets a
|
|
323
|
+
zero-width span where its text would start.
|
|
324
|
+
- Only top-level bits get spans. For any other output (`lex`, `ast`,
|
|
325
|
+
`semantic-tokens`, `diagnostics`, a mapping report) `bitSpans` is present
|
|
326
|
+
with `spans: []`; without the flag it is absent.
|
|
327
|
+
|
|
328
|
+
The CLI's `--bit-spans FILE` and the daemon's `convert` option
|
|
329
|
+
`bitSpans: true` (reply field `bitSpans`) return the same object.
|
|
330
|
+
|
|
289
331
|
#### canonicalize(input: string, options?: CanonicalizeOptions): string
|
|
290
332
|
|
|
291
333
|
Re-emit input in its own format in canonical form — the single same-format surface. `mode`: `"optimized"` (default, omit natural defaults) or `"full"` (all keys).
|
|
@@ -449,7 +491,14 @@ included so the alignment is complete — `status`
|
|
|
449
491
|
`ops` (a patch turning the A bit into the B bit) and `revert`, the parser
|
|
450
492
|
issue codes gained or lost, and the rendered `hunks`.
|
|
451
493
|
|
|
452
|
-
#### countBits(input: string): number / splitBits(input: string): BitSlice[]
|
|
494
|
+
#### countBits(input: string): number / splitBits(input: string, options?: SplitBitsOptions): BitSlice[]
|
|
495
|
+
|
|
496
|
+
`splitBits` returns one `{ index, start, end, byteStart, byteEnd, source }`
|
|
497
|
+
per bit. `start` / `end` locate the bit's text, `source`, counted in
|
|
498
|
+
`options.positionEncoding` — `"utf-16"` by default, so
|
|
499
|
+
`input.slice(start, end) === source` — or `"utf-8"` bytes. `byteStart` /
|
|
500
|
+
`byteEnd` are UTF-8 byte offsets that tile the document (each bit runs to the
|
|
501
|
+
next bit's start), for chunking.
|
|
453
502
|
|
|
454
503
|
Count or split the bits of a bitmark/JSON input without full parsing. A JSON
|
|
455
504
|
input may be an array or a single value, each entry a `{ bit: {…} }` envelope
|
|
@@ -1086,6 +1135,10 @@ bitmark convert input.bitmark --output-format semantic-tokens --tokens-layout ab
|
|
|
1086
1135
|
# Dump the lexer token stream as JSON (debug tooling; bitmark input only)
|
|
1087
1136
|
bitmark convert input.bitmark --input-format bitmark --output-format lex --pretty
|
|
1088
1137
|
|
|
1138
|
+
# Also write where each bit landed in the output, {positionEncoding, spans},
|
|
1139
|
+
# to its own file (offsets into this run's output; --position-encoding applies)
|
|
1140
|
+
bitmark convert input.bitmark --output-format html -o out.html --bit-spans spans.json
|
|
1141
|
+
|
|
1089
1142
|
# Breakscape / unbreakscape text
|
|
1090
1143
|
bitmark breakscape input.txt
|
|
1091
1144
|
bitmark breakscape input.txt --format plainText --location tag
|