@preventive/triage 1.0.0-alpha.15 → 1.0.0-alpha.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,13 @@
1
+ export const DEFAULT_SCAN_SERVER = 'http://127.0.0.1:3123/'
2
+
3
+ // Discovery is optional: malformed scan configuration must never prevent
4
+ // opening local data or recognizing the server's sync protocol.
5
+ export function normalizeScanServer(value: unknown): string | null {
6
+ if (typeof value !== 'string' || !value.trim()) return null
7
+ try {
8
+ const url = new URL(value)
9
+ if (!['http:', 'https:'].includes(url.protocol) || url.username || url.password || url.search || url.hash) return null
10
+ url.pathname = url.pathname.replace(/\/+$/u, '') + '/'
11
+ return url.href
12
+ } catch { return null }
13
+ }
@@ -22,6 +22,7 @@ export interface ManagedServerInfo {
22
22
  export interface ServerInfo {
23
23
  mode: ServerMode
24
24
  managed: ManagedServerInfo | null
25
+ deepviewScanServer?: string
25
26
  }
26
27
 
27
28
  // The mode-probe route. A client GETs this to learn a server's protocol up
package/common/utf8.d.ts CHANGED
@@ -11,3 +11,6 @@
11
11
  // narrower declaration is safe.
12
12
  export function encodeUtf8(str: string): Uint8Array<ArrayBuffer>
13
13
  export function decodeUtf8(bytes: Uint8Array | ArrayBuffer | ArrayBufferView): string
14
+ // Bytes `str` takes in UTF-8, measured without encoding when no character
15
+ // is past \xFF. A lone surrogate counts as the U+FFFD it encodes to.
16
+ export function utf8ByteLength(str: string): number
package/common/utf8.js CHANGED
@@ -43,6 +43,51 @@ export function encodeUtf8(str) {
43
43
  return encoder.encode(str)
44
44
  }
45
45
 
46
+ // ASCII characters in a row after which the count hands back to the
47
+ // native scan. A jump costs a call and a step does not, so it pays only
48
+ // across a real gap: at 1, alternating text (`aéaé…`) takes a jump per
49
+ // character, and past 2, text with an accent every few words steps
50
+ // through ASCII the scan crosses faster. Measured on 50 MB of each, 2 was
51
+ // never more than 1.5x the best of the thresholds tried, 1 to 32.
52
+ const ASCII_RUN = 2
53
+
54
+ // The number of bytes `str` takes in UTF-8, without encoding it where
55
+ // that can be helped. A string with no character past \xFF is Latin-1:
56
+ // ASCII takes a byte and \x80-\xFF two, so its size is its length plus
57
+ // its high characters. An engine stores such a string a byte per
58
+ // character, where no character past \xFF can be, so the first test is
59
+ // answered without reading it. Only a string with a wider character is
60
+ // encoded to be measured.
61
+ //
62
+ // The high characters are counted without collecting them: a match array
63
+ // takes an entry apiece, which for Latin-1-heavy text outweighs the
64
+ // encoding this avoids. A native scan jumps to the next one — for ASCII
65
+ // it finds none — and a loop counts on from there while they keep coming,
66
+ // handing back to the scan after a run of ASCII. Sparse text is read
67
+ // natively, dense text by the loop, and neither allocates.
68
+ //
69
+ // This sizes text for display, so unlike `encodeUtf8` it does not refuse
70
+ // a lone surrogate: it counts the U+FFFD that TextEncoder writes for one.
71
+ export function utf8ByteLength(str) {
72
+ if (typeof str !== 'string') {
73
+ throw new TypeError(`utf8ByteLength expects a string, got ${typeof str}`)
74
+ }
75
+ // Code units, not code points: a `u` regexp reads a two-byte string by
76
+ // code point, several times slower, and a surrogate is past \xFF either way.
77
+ // eslint-disable-next-line require-unicode-regexp
78
+ if (/[\u0100-\uFFFF]/.test(str)) return encoder.encode(str).byteLength
79
+ const high = /[\u0080-\u00FF]/gu
80
+ let bytes = str.length
81
+ while (high.test(str)) {
82
+ let i = high.lastIndex - 1
83
+ for (let ascii = 0; i < str.length && ascii < ASCII_RUN; i++) {
84
+ if (str.codePointAt(i) > 0x7F) { bytes++; ascii = 0 } else ascii++
85
+ }
86
+ high.lastIndex = i
87
+ }
88
+ return bytes
89
+ }
90
+
46
91
  export function decodeUtf8(bytes) {
47
92
  // Reject non-BufferSource so a missed destructure / optional field /
48
93
  // misnamed property surfaces here instead of silently defaulting to