@wafertools/testdata-parser 0.10.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +175 -30
- package/package.json +1 -1
- package/testdata_parser.d.ts +214 -13
- package/testdata_parser.js +29 -18
- package/testdata_parser_bg.wasm +0 -0
package/README.md
CHANGED
|
@@ -21,16 +21,55 @@ import init, { parse_stdf } from '@wafertools/testdata-parser';
|
|
|
21
21
|
|
|
22
22
|
await init(); // fetches testdata_parser_bg.wasm relative to the module URL
|
|
23
23
|
const bytes = new Uint8Array(await file.arrayBuffer());
|
|
24
|
-
const parsed = parse_stdf(bytes); // ParsedStdf, or throws a
|
|
24
|
+
const parsed = parse_stdf(bytes); // ParsedStdf, or throws a ParserError
|
|
25
25
|
```
|
|
26
26
|
|
|
27
|
+
**Import `init` as the default export, not by name.** There is also a named `init`
|
|
28
|
+
export — that one is only the panic hook, and awaiting it instantiates nothing. The
|
|
29
|
+
default export is what loads the WASM binary.
|
|
30
|
+
|
|
31
|
+
**Failures throw a `ParserError`**: a real `Error`, so `err.message` reads and
|
|
32
|
+
`err instanceof Error` is true, carrying a stable `err.code` to branch on. Handle the
|
|
33
|
+
code, display the message — messages are prose and may be reworded.
|
|
34
|
+
|
|
35
|
+
```js
|
|
36
|
+
try {
|
|
37
|
+
const parsed = parse_stdf(bytes);
|
|
38
|
+
} catch (err) {
|
|
39
|
+
if (err.code === 'not-stdf') { // the file is not the format its name claims
|
|
40
|
+
// ...offer to try another parser
|
|
41
|
+
}
|
|
42
|
+
showToast(err.message);
|
|
43
|
+
}
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The full set: `file-read`, `gzip-invalid`, `encoding-invalid`, `not-stdf`,
|
|
47
|
+
`stdf-unsupported`, `csv-read`, `json-invalid`, `parquet-read`, `column-missing`,
|
|
48
|
+
`mapping-invalid`, `internal`. `ParseErrorCode` in the type declarations is the same
|
|
49
|
+
list, so a `switch` on it is exhaustively checked.
|
|
50
|
+
|
|
27
51
|
`init()` also installs a panic hook that routes any Rust panic to `console.error` with a stack trace, instead of an opaque WASM trap.
|
|
28
52
|
|
|
29
53
|
In a bundler/dev-server context, `new URL('...testdata_parser_bg.wasm', import.meta.url)` resolution can be finicky — see tsmap's `parserWorker.ts` for a worked example of loading this module off the main thread in a Vite app.
|
|
30
54
|
|
|
55
|
+
## TypeScript
|
|
56
|
+
|
|
57
|
+
The package ships real declarations for every result and input shape —
|
|
58
|
+
`ParsedStdf`, `WaferData`, `DieResult`, `TestDef`, `ParserWarning`, `ScanResult`,
|
|
59
|
+
`FileMeta`, `ParquetHeadersResult`, `CsvMapping`, and the `ParserWarningCode` /
|
|
60
|
+
`ParseErrorCode` unions. Each export is typed with its real return type, so field
|
|
61
|
+
names complete and a typo is a compile error.
|
|
62
|
+
|
|
63
|
+
They are emitted from the Rust (`typescript_custom_section` in `lib.rs`) into
|
|
64
|
+
`testdata_parser.d.ts`, so they ship with the package and match the version
|
|
65
|
+
installed. Earlier versions typed every return value as `any`, which is why the type
|
|
66
|
+
blocks below existed as the only contract — they are now a readable mirror of the
|
|
67
|
+
declarations, and `scripts/check-parser-docs.mjs` holds the Rust, the declarations
|
|
68
|
+
and this README to each other.
|
|
69
|
+
|
|
31
70
|
## API
|
|
32
71
|
|
|
33
|
-
Every parse function takes raw file bytes (`Uint8Array`) and returns a plain JS object (via `serde-wasm-bindgen`), or throws a `
|
|
72
|
+
Every parse function takes raw file bytes (`Uint8Array`) and returns a plain JS object (via `serde-wasm-bindgen`), or throws a `ParserError` (see above). Gzip-compressed input (`.gz`) is transparently decompressed for every format.
|
|
34
73
|
|
|
35
74
|
| Function | Signature | Returns |
|
|
36
75
|
| --- | --- | --- |
|
|
@@ -42,6 +81,9 @@ Every parse function takes raw file bytes (`Uint8Array`) and returns a plain JS
|
|
|
42
81
|
| `parse_parquet` | `(bytes: Uint8Array, mapping: CsvMapping) => ParsedStdf` | Full parse of a Parquet file, using the same mapping shape as CSV/JSON |
|
|
43
82
|
| `stdf_test_names` | `(bytes: Uint8Array) => ScanResult` | Fast first-pass scan: test definitions + die count, no die accumulation |
|
|
44
83
|
| `atdf_test_names` | `(bytes: Uint8Array) => ScanResult` | Same first-pass scan for ATDF |
|
|
84
|
+
| `stdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Lot metadata, wafer count, first/last timestamps and site count from an MIR/SDR/WIR/WRR-only scan — no PTR/FTR/PIR/PRR walk, so it stays cheap across a batch of files |
|
|
85
|
+
| `atdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Same metadata-only scan for ATDF |
|
|
86
|
+
| `parquet_distinct_count` | `(bytes: Uint8Array, columns: string[]) => number` | How many distinct combinations of those columns the file holds — a wafer count from `['lot','wafer']` without a full parse, read as a column projection. A column missing from the schema is an error, not a count of zero. Parquet only: CSV/JSON have no equivalent shortcut |
|
|
45
87
|
| `parse_stdf_filtered` | `(bytes: Uint8Array, selected: number[]) => ParsedStdf` | Full parse, skipping per-site accumulation for test numbers not in `selected` |
|
|
46
88
|
| `parse_atdf_filtered` | `(bytes: Uint8Array, selected: number[]) => ParsedStdf` | Same filtered parse for ATDF |
|
|
47
89
|
|
|
@@ -61,8 +103,8 @@ STDF and ATDF files can be large and contain far more tests than a caller wants
|
|
|
61
103
|
|
|
62
104
|
```ts
|
|
63
105
|
interface CsvMapping {
|
|
64
|
-
x
|
|
65
|
-
y
|
|
106
|
+
x?: string | null; // die X coordinate column — omit for data with no positions
|
|
107
|
+
y?: string | null; // die Y coordinate column — see `x`
|
|
66
108
|
hbin?: string; // hardware bin column
|
|
67
109
|
sbin?: string; // software bin column
|
|
68
110
|
wafer?: string; // wafer ID column (groups rows into WaferData[])
|
|
@@ -87,6 +129,11 @@ interface CsvTestCol {
|
|
|
87
129
|
}
|
|
88
130
|
```
|
|
89
131
|
|
|
132
|
+
`tests`, `meta`, `splitBy` and `passBins` must be present — pass `[]` where you are not
|
|
133
|
+
using one. Every other field may be omitted entirely or set to `null`; the two mean the
|
|
134
|
+
same thing. (`parse_csv::mapping_shape_tests` pins that split, and the `CsvMapping`
|
|
135
|
+
declaration shipped in the package encodes it.)
|
|
136
|
+
|
|
90
137
|
Two ways to describe test columns are supported: a **fixed set** of `tests` (one column per test, "wide" format), or a **tall** layout (`testnameCol`/`testnumberCol`/`testvalueCol` — one row per die×test, with the test identity read from a column rather than the header).
|
|
91
138
|
|
|
92
139
|
**Test identity — real number vs. synthesized one.** Neither format has a mandatory real STDF-style test number, so one gets synthesized by default (see "Design notes" below) — but a caller that *does* have real numbers in the source data shouldn't lose them:
|
|
@@ -102,7 +149,10 @@ interface ParsedStdf {
|
|
|
102
149
|
wafers: WaferData[];
|
|
103
150
|
testDefs: Record<string, TestDef>; // keyed by test number as a string
|
|
104
151
|
sites: SiteInfo[];
|
|
105
|
-
|
|
152
|
+
hbinDefs?: BinDef[]; // hard-bin names from HBR; omitted if empty
|
|
153
|
+
sbinDefs?: BinDef[]; // soft-bin names from SBR; omitted if empty
|
|
154
|
+
passHbins?: number[]; // hard bins HBR marks Pass (HBIN_PF == 'P'); omitted if empty
|
|
155
|
+
warnings?: ParserWarning[]; // non-fatal advisories; omitted if empty
|
|
106
156
|
}
|
|
107
157
|
|
|
108
158
|
interface WaferData {
|
|
@@ -115,13 +165,15 @@ interface WaferData {
|
|
|
115
165
|
}
|
|
116
166
|
|
|
117
167
|
interface DieResult {
|
|
118
|
-
x
|
|
119
|
-
y
|
|
168
|
+
x?: number; // die grid X — absent on an unpositioned die, see below
|
|
169
|
+
y?: number; // die grid Y — absent on an unpositioned die, see below
|
|
170
|
+
dieIndex?: number; // per-wafer ordinal, present ONLY when x/y are absent
|
|
120
171
|
hbin?: number;
|
|
121
172
|
sbin?: number;
|
|
122
173
|
siteNum?: number;
|
|
123
174
|
partId?: number;
|
|
124
|
-
testValues?: Record<string, number>;
|
|
175
|
+
testValues?: Record<string, number>; // keyed by test number as a string
|
|
176
|
+
testPass?: Record<string, boolean>; // recorded verdicts, true = pass; see below
|
|
125
177
|
}
|
|
126
178
|
|
|
127
179
|
interface TestDef {
|
|
@@ -146,8 +198,63 @@ interface SiteInfo {
|
|
|
146
198
|
headNum: number;
|
|
147
199
|
siteNum: number;
|
|
148
200
|
}
|
|
201
|
+
|
|
202
|
+
interface BinDef {
|
|
203
|
+
bin: number;
|
|
204
|
+
name: string;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
interface ParserWarning {
|
|
208
|
+
code: ParserWarningCode;
|
|
209
|
+
message: string;
|
|
210
|
+
severity: 'warning' | 'error';
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
type ParserWarningCode =
|
|
214
|
+
| 'unpositioned-dies' // dies with no X/Y — real data, not placeable
|
|
215
|
+
| 'soft-bin-mirrored' // sbin was the 65535 sentinel; hbin mirrored in
|
|
216
|
+
| 'values-not-numeric' // a mapped column held values that would not coerce
|
|
217
|
+
| 'retests-assumed' // repeated positions read as retests
|
|
218
|
+
| 'wafer-split-by-column' // one wafer per value of a mapped column
|
|
219
|
+
| 'column-varies-within-wafer'// a metadata column describes dies, not wafers
|
|
220
|
+
| 'multiple-lot-records'; // the file holds more than one MIR
|
|
149
221
|
```
|
|
150
222
|
|
|
223
|
+
**`x`/`y` are optional, and a die is either fully positioned or fully unpositioned — never
|
|
224
|
+
half.** A die with no reported position has neither field and carries `dieIndex` instead (its
|
|
225
|
+
PRR/row encounter order within the wafer), giving it a stable identity that survives
|
|
226
|
+
filtering and sorting downstream. Such a die still holds real measured data and counts
|
|
227
|
+
toward every non-spatial statistic, but cannot be placed on a wafer map. Do not default a
|
|
228
|
+
missing coordinate to 0 — that invents a die at the origin. Each wafer holding any of them
|
|
229
|
+
also produces a `warnings` entry naming how many.
|
|
230
|
+
|
|
231
|
+
**`testPass` holds recorded pass/fail verdicts**, keyed exactly like `testValues`, `true`
|
|
232
|
+
meaning pass. Functional (FTR) results live here and *only* here — they have no measured
|
|
233
|
+
value, so they never appear in `testValues`. A parametric (PTR) test also gets an entry when
|
|
234
|
+
the tester recorded a valid indication (STDF `TEST_FLG` bit 6 clear). A test number absent
|
|
235
|
+
from the map has no recorded verdict, which is not the same as a fail.
|
|
236
|
+
|
|
237
|
+
**`hbinDefs`/`sbinDefs` carry bin names from HBR/SBR records**, one entry per distinct bin
|
|
238
|
+
number that had a non-empty name — a bin with no recorded name is omitted rather than
|
|
239
|
+
emitted with an empty string, so a host can fall back to its own "Bin N" label. Hard and
|
|
240
|
+
soft bins occupy independent number spaces (STDF V4), which is why they are two arrays and
|
|
241
|
+
never merged. Entries are sorted by bin number.
|
|
242
|
+
|
|
243
|
+
**`passHbins` is the file's own pass/fail truth** — the hard bins whose HBR marked
|
|
244
|
+
`HBIN_PF == 'P'`. It matters because a consumer that does not carry it across generally
|
|
245
|
+
assumes bin 1 is the pass bin, which decides both the yield number and how it is labelled
|
|
246
|
+
whether or not it is true of this test program. It is absent when no HBR record carried a
|
|
247
|
+
usable Pass flag; in that case leave the consumer's own default in place rather than passing
|
|
248
|
+
an empty array, which asserts that nothing passes.
|
|
249
|
+
|
|
250
|
+
**`warnings` carries a stable `code`, prose, and a severity** — branch on the code, display
|
|
251
|
+
the message, and never match on the prose. `severity: 'error'` means a number or a plot
|
|
252
|
+
built from this result can mislead, because data was dropped or a value was substituted
|
|
253
|
+
(`unpositioned-dies`, `soft-bin-mirrored`, `values-not-numeric`); `'warning'` means the
|
|
254
|
+
parse made a documented interpretation you may want to change, and nothing was altered or
|
|
255
|
+
lost. Nothing here is fatal — the parse succeeded. Surface them: a silently discarded
|
|
256
|
+
warning is how a partly-wrong load looks fine.
|
|
257
|
+
|
|
151
258
|
`testDefs` and `testValues` are both keyed by **test number**, not test name — test numbers are the unique identity in STDF/ATDF; names are not guaranteed unique.
|
|
152
259
|
|
|
153
260
|
`meta`/`fields` are intentionally generic key/value pairs rather than a fixed struct: new metadata fields flow through from the source format with no type or crate change, and it's up to the host application to decide which fields to surface and how to label them.
|
|
@@ -161,6 +268,26 @@ interface ScanResult {
|
|
|
161
268
|
}
|
|
162
269
|
```
|
|
163
270
|
|
|
271
|
+
### Return shape — `FileMeta`
|
|
272
|
+
|
|
273
|
+
Returned by `stdf_file_meta`/`atdf_file_meta`: enough to list or filter a batch of files
|
|
274
|
+
without parsing any of them fully. One `FileMeta` describes one *file*, so the timestamps
|
|
275
|
+
and wafer count are aggregated across every WIR/WRR pair in it.
|
|
276
|
+
|
|
277
|
+
```ts
|
|
278
|
+
interface FileMeta {
|
|
279
|
+
lotMeta: LotMeta;
|
|
280
|
+
waferCount: number;
|
|
281
|
+
earliestStart?: string; // earliest WIR START_T; absent if the file has no WIR records
|
|
282
|
+
latestFinish?: string; // latest WRR FINISH_T; absent if no wafer completed
|
|
283
|
+
siteCount?: number; // distinct site numbers across every SDR seen
|
|
284
|
+
}
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
Timestamps are normally fixed-width ISO 8601, which is what makes "earliest" a plain string
|
|
288
|
+
compare — but an ATDF value written in some other convention passes through unrecognised
|
|
289
|
+
rather than being dropped, so treat these as display text unless you have parsed them.
|
|
290
|
+
|
|
164
291
|
### Column headers: CSV/JSON vs Parquet
|
|
165
292
|
|
|
166
293
|
There is no `csv_headers`/`json_headers` in the WASM API — a browser caller that needs to show the user a column-mapping UI before parsing a CSV/JSON file can just read the header row itself in plain JS (this is what tsmap's web build does; the desktop build calls the native functions below instead). The byte-based Rust functions exist (`csv_headers_from_bytes`, and `json_headers_sync`'s logic), they are simply not wired through `wasm-bindgen` for these two formats.
|
|
@@ -184,33 +311,41 @@ The crate also builds as a native Rust library (used directly by tsmap's Tauri c
|
|
|
184
311
|
|
|
185
312
|
| Function | Module |
|
|
186
313
|
| --- | --- |
|
|
187
|
-
| `parse_stdf_sync(path: String) ->
|
|
188
|
-
| `parse_atdf_sync(path: String) ->
|
|
189
|
-
| `csv_headers_inner(path: String) ->
|
|
190
|
-
| `parse_csv_inner(path: String, mapping: CsvMapping) ->
|
|
191
|
-
| `json_headers_sync(path: String) ->
|
|
192
|
-
| `parse_json_sync(path: String, mapping: CsvMapping) ->
|
|
193
|
-
| `parquet_headers_inner(path: String) ->
|
|
194
|
-
| `parse_parquet_inner(path: String, mapping: CsvMapping) ->
|
|
195
|
-
| `
|
|
196
|
-
| `
|
|
314
|
+
| `parse_stdf_sync(path: String) -> ParseResult<ParsedStdf>` | `parse_stdf` |
|
|
315
|
+
| `parse_atdf_sync(path: String) -> ParseResult<ParsedStdf>` | `parse_atdf` |
|
|
316
|
+
| `csv_headers_inner(path: String) -> ParseResult<CsvHeadersResult>` | `parse_csv` |
|
|
317
|
+
| `parse_csv_inner(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_csv` |
|
|
318
|
+
| `json_headers_sync(path: String) -> ParseResult<JsonHeadersResult>` | `parse_json` |
|
|
319
|
+
| `parse_json_sync(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
|
|
320
|
+
| `parquet_headers_inner(path: String) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
|
|
321
|
+
| `parse_parquet_inner(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
|
|
322
|
+
| `parquet_distinct_count_inner(path: String, columns: Vec<String>) -> ParseResult<usize>` | `parse_parquet` |
|
|
323
|
+
| `read_bytes(path: &str) -> ParseResult<Vec<u8>>` | `read_file` |
|
|
324
|
+
| `read_text(path: &str) -> ParseResult<String>` | `read_file` |
|
|
325
|
+
|
|
326
|
+
Every fallible function returns `ParseResult<T>` — that is `Result<T, ParseError>`, where
|
|
327
|
+
`ParseError` is `{ code: &'static str, message: String }` and `Display` renders the message
|
|
328
|
+
alone. Same codes as the WASM layer above; it is the same error, serialised there.
|
|
197
329
|
|
|
198
330
|
**Byte-based** — available on every target, and what the WASM exports wrap. Use these from Rust when you already hold the bytes:
|
|
199
331
|
|
|
200
332
|
| Function | Module |
|
|
201
333
|
| --- | --- |
|
|
202
|
-
| `parse_stdf_from_bytes(&[u8]) ->
|
|
203
|
-
| `parse_atdf_from_bytes(&[u8]) ->
|
|
204
|
-
| `parse_stdf_test_names(&[u8]) ->
|
|
205
|
-
| `parse_atdf_test_names(&[u8]) ->
|
|
206
|
-
| `parse_stdf_from_bytes_filtered(&[u8], &HashSet<u32>) ->
|
|
207
|
-
| `parse_atdf_from_bytes_filtered(&[u8], &HashSet<u32>) ->
|
|
208
|
-
| `
|
|
209
|
-
| `
|
|
210
|
-
| `
|
|
211
|
-
| `
|
|
212
|
-
| `
|
|
213
|
-
| `
|
|
334
|
+
| `parse_stdf_from_bytes(&[u8]) -> ParseResult<ParsedStdf>` | `parse_stdf` |
|
|
335
|
+
| `parse_atdf_from_bytes(&[u8]) -> ParseResult<ParsedStdf>` | `parse_atdf` |
|
|
336
|
+
| `parse_stdf_test_names(&[u8]) -> ParseResult<ScanResult>` | `parse_stdf` |
|
|
337
|
+
| `parse_atdf_test_names(&[u8]) -> ParseResult<ScanResult>` | `parse_atdf` |
|
|
338
|
+
| `parse_stdf_from_bytes_filtered(&[u8], &HashSet<u32>) -> ParseResult<ParsedStdf>` | `parse_stdf` |
|
|
339
|
+
| `parse_atdf_from_bytes_filtered(&[u8], &HashSet<u32>) -> ParseResult<ParsedStdf>` | `parse_atdf` |
|
|
340
|
+
| `parse_stdf_file_meta(&[u8]) -> Result<FileMeta, String>` | `parse_stdf` |
|
|
341
|
+
| `parse_atdf_file_meta(&[u8]) -> Result<FileMeta, String>` | `parse_atdf` |
|
|
342
|
+
| `csv_headers_from_bytes(&[u8]) -> ParseResult<CsvHeadersResult>` | `parse_csv` |
|
|
343
|
+
| `parse_csv_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_csv` |
|
|
344
|
+
| `parse_json_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
|
|
345
|
+
| `parquet_headers_from_bytes(&[u8]) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
|
|
346
|
+
| `parse_parquet_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
|
|
347
|
+
| `parquet_distinct_count_from_bytes(&[u8], &[String]) -> ParseResult<usize>` | `parse_parquet` |
|
|
348
|
+
| `decompress_if_gzip(Vec<u8>) -> ParseResult<Vec<u8>>` | `read_file` |
|
|
214
349
|
|
|
215
350
|
`CsvHeadersResult` and `JsonHeadersResult` are the same shape — the header row plus enough of the file to preview a mapping:
|
|
216
351
|
|
|
@@ -231,6 +366,16 @@ pub struct CsvHeadersResult {
|
|
|
231
366
|
- **Parquet reads through a row-oriented API, not Arrow.** `parquet::record::Row`/`Field` rather than the `arrow` feature — a closer fit for this crate's row-based `DieResult` model, and a smaller WASM bundle (no Arrow array machinery pulled in). A typed Parquet cell is coerced to `f64` for numeric roles and to a plain string otherwise; a value that fails to coerce (e.g. a numeric role mapped to a genuinely string-typed column) is skipped and surfaced as one summarised entry in `warnings`, not a panic or a silent zero.
|
|
232
367
|
- **Parquet's `zstd` codec is native-only.** `snappy`, `gzip`, `lz4`, and `brotli` build for `wasm32-unknown-unknown` with no extra toolchain; `zstd`'s C library needs a real C cross-compiler targeting wasm32, which a plain `wasm-pack build` doesn't assume is available. A `zstd`-compressed Parquet file parses natively but fails clearly on the WASM build.
|
|
233
368
|
|
|
369
|
+
## Using this package with an AI coding agent
|
|
370
|
+
|
|
371
|
+
`llms.txt` ships with the package (`node_modules/@wafertools/testdata-parser/llms.txt`) and
|
|
372
|
+
is written to be handed to a coding agent: a map of the entry points, the traps that produce
|
|
373
|
+
a silently wrong parse, and how the result hands off to
|
|
374
|
+
[`@wafertools/wafermap`](https://wafertools.github.io/wafermap/). It is worth pointing an
|
|
375
|
+
agent at, because `testdata_parser.d.ts` is wasm-bindgen output and types every return value
|
|
376
|
+
as `any` — the result shapes above are the only contract, and an agent that has not read them
|
|
377
|
+
will invent field names.
|
|
378
|
+
|
|
234
379
|
## Versioning
|
|
235
380
|
|
|
236
381
|
This crate has its own release lifecycle, independent of any consuming application's version. See the parent repository's `CLAUDE.md` for the bump/build/publish steps.
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "@wafertools/testdata-parser",
|
|
3
3
|
"type": "module",
|
|
4
4
|
"description": "Rust/WASM parsers for semiconductor test data formats (STDF, ATDF, CSV, JSON, Parquet)",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.11.0",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"repository": {
|
|
8
8
|
"type": "git",
|
package/testdata_parser.d.ts
CHANGED
|
@@ -1,36 +1,237 @@
|
|
|
1
1
|
/* tslint:disable */
|
|
2
2
|
/* eslint-disable */
|
|
3
3
|
|
|
4
|
-
|
|
4
|
+
/** One metadata field, exactly as the source file recorded it. */
|
|
5
|
+
export interface MetaField {
|
|
6
|
+
key: string;
|
|
7
|
+
value: string;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/** Lot-level metadata: every non-empty field of the source's lot record (MIR). */
|
|
11
|
+
export interface LotMeta {
|
|
12
|
+
fields: MetaField[];
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export interface SiteInfo {
|
|
16
|
+
headNum: number;
|
|
17
|
+
siteNum: number;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** One bin's name, from an HBR (hard) or SBR (soft) record. */
|
|
21
|
+
export interface BinDef {
|
|
22
|
+
bin: number;
|
|
23
|
+
name: string;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface TestDef {
|
|
27
|
+
name: string;
|
|
28
|
+
/** "P" parametric (has a measured value) or "F" functional (verdict only). */
|
|
29
|
+
testType: "P" | "F";
|
|
30
|
+
loLimit?: number;
|
|
31
|
+
hiLimit?: number;
|
|
32
|
+
units?: string;
|
|
33
|
+
/** The file's own display order. Absent for STDF/ATDF, where the real test
|
|
34
|
+
* number already sorts meaningfully. */
|
|
35
|
+
order?: number;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface DieResult {
|
|
39
|
+
/** Die grid X. Absent — together with `y` — on a die the file gave no
|
|
40
|
+
* position for; never one without the other. Do not default it to 0. */
|
|
41
|
+
x?: number;
|
|
42
|
+
/** Die grid Y. See `x`. */
|
|
43
|
+
y?: number;
|
|
44
|
+
/** Per-wafer ordinal, present only when `x`/`y` are absent, so an
|
|
45
|
+
* unpositioned die still has a stable identity. */
|
|
46
|
+
dieIndex?: number;
|
|
47
|
+
hbin?: number;
|
|
48
|
+
sbin?: number;
|
|
49
|
+
siteNum?: number;
|
|
50
|
+
partId?: number;
|
|
51
|
+
/** Measured values, keyed by test number as a string. Parametric tests only. */
|
|
52
|
+
testValues?: Record<string, number>;
|
|
53
|
+
/** Recorded pass/fail verdicts, `true` = pass, keyed like `testValues`.
|
|
54
|
+
* Functional results live here and only here. A test absent from this map has
|
|
55
|
+
* no recorded verdict, which is not a fail. */
|
|
56
|
+
testPass?: Record<string, boolean>;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export interface WaferData {
|
|
60
|
+
waferId: string;
|
|
61
|
+
results: DieResult[];
|
|
62
|
+
partCount?: number;
|
|
63
|
+
goodCount?: number;
|
|
64
|
+
failCount?: number;
|
|
65
|
+
/** Per-wafer metadata (WIR/WRR). Absent for formats with no wafer records. */
|
|
66
|
+
fields?: MetaField[];
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** Every advisory a parse can raise. Branch on this, never on the message. */
|
|
70
|
+
export type ParserWarningCode =
|
|
71
|
+
| "unpositioned-dies"
|
|
72
|
+
| "soft-bin-mirrored"
|
|
73
|
+
| "values-not-numeric"
|
|
74
|
+
| "retests-assumed"
|
|
75
|
+
| "wafer-split-by-column"
|
|
76
|
+
| "column-varies-within-wafer"
|
|
77
|
+
| "multiple-lot-records";
|
|
78
|
+
|
|
79
|
+
/** A non-fatal advisory. The parse succeeded; something about it is worth
|
|
80
|
+
* knowing. `"error"` means a number or plot built from this result can mislead,
|
|
81
|
+
* because data was dropped or a value substituted; `"warning"` means the parse
|
|
82
|
+
* made a documented interpretation you may want to change. */
|
|
83
|
+
export interface ParserWarning {
|
|
84
|
+
code: ParserWarningCode;
|
|
85
|
+
message: string;
|
|
86
|
+
severity: "warning" | "error";
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** What every format parses to. There is no format-specific result type. */
|
|
90
|
+
export interface ParsedStdf {
|
|
91
|
+
meta: LotMeta;
|
|
92
|
+
wafers: WaferData[];
|
|
93
|
+
/** Keyed by test number as a string. */
|
|
94
|
+
testDefs: Record<string, TestDef>;
|
|
95
|
+
sites: SiteInfo[];
|
|
96
|
+
/** Hard-bin names from HBR. Absent when the file named no bins. */
|
|
97
|
+
hbinDefs?: BinDef[];
|
|
98
|
+
/** Soft-bin names from SBR. Absent when the file named no bins. */
|
|
99
|
+
sbinDefs?: BinDef[];
|
|
100
|
+
/** Hard bins the file marks as Pass. Absent when no HBR carried the flag —
|
|
101
|
+
* then leave your consumer's own default alone rather than passing `[]`,
|
|
102
|
+
* which asserts that nothing passes. */
|
|
103
|
+
passHbins?: number[];
|
|
104
|
+
/** Absent when there is nothing to report. */
|
|
105
|
+
warnings?: ParserWarning[];
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** First-pass scan: what tests the file holds, and how many dies. */
|
|
109
|
+
export interface ScanResult {
|
|
110
|
+
testDefs: Record<string, TestDef>;
|
|
111
|
+
dieCount: number;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** Metadata-only scan of one file, for listing or filtering a batch. */
|
|
115
|
+
export interface FileMeta {
|
|
116
|
+
lotMeta: LotMeta;
|
|
117
|
+
waferCount: number;
|
|
118
|
+
/** Earliest WIR START_T. Normally ISO 8601, but an ATDF value in another
|
|
119
|
+
* convention passes through as-is — display text unless you parse it. */
|
|
120
|
+
earliestStart?: string;
|
|
121
|
+
/** Latest WRR FINISH_T, same caveat. */
|
|
122
|
+
latestFinish?: string;
|
|
123
|
+
siteCount?: number;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
export interface ParquetHeadersResult {
|
|
127
|
+
headers: string[];
|
|
128
|
+
sample: Record<string, string>[];
|
|
129
|
+
rowCount: number;
|
|
130
|
+
/** Coarse per-column kind, inferred from the first sampled row. */
|
|
131
|
+
columnTypes: Record<string, "number" | "bool" | "string">;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** One test column in a wide-format mapping. */
|
|
135
|
+
export interface CsvTestCol {
|
|
136
|
+
col: string;
|
|
137
|
+
/** Assigned by you, not by the parser. A column whose header is itself a
|
|
138
|
+
* number should use that number. */
|
|
139
|
+
testNumber: number;
|
|
140
|
+
name: string;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Column mapping for CSV, JSON and Parquet. There is no header
|
|
144
|
+
* auto-detection, so this is required for those three formats.
|
|
145
|
+
*
|
|
146
|
+
* `tests`, `meta`, `splitBy` and `passBins` must be present — pass `[]` for
|
|
147
|
+
* the ones you are not using. Every other field may be omitted or set to
|
|
148
|
+
* `null`, which mean the same thing. */
|
|
149
|
+
export interface CsvMapping {
|
|
150
|
+
/** Die X column. Omit — together with `y` — for data with no positions. */
|
|
151
|
+
x?: string | null;
|
|
152
|
+
/** Die Y column. See `x`. */
|
|
153
|
+
y?: string | null;
|
|
154
|
+
hbin?: string | null;
|
|
155
|
+
sbin?: string | null;
|
|
156
|
+
/** Groups rows into wafers. */
|
|
157
|
+
wafer?: string | null;
|
|
158
|
+
lot?: string | null;
|
|
159
|
+
/** Test site number. Non-numeric values mean no site for that die. */
|
|
160
|
+
site?: string | null;
|
|
161
|
+
/** Wide format: one entry per test-value column. `[]` for tall format. */
|
|
162
|
+
tests: CsvTestCol[];
|
|
163
|
+
/** Extra columns to carry through as per-die metadata. */
|
|
164
|
+
meta: string[];
|
|
165
|
+
/** Columns to facet wafers by, beyond `wafer`. */
|
|
166
|
+
splitBy: string[];
|
|
167
|
+
/** Tall format: the column holding each row's test name. */
|
|
168
|
+
testnameCol?: string | null;
|
|
169
|
+
/** Tall format: the column holding each row's real test number. Set this
|
|
170
|
+
* where the data has one — otherwise a number is hashed from the name. */
|
|
171
|
+
testnumberCol?: string | null;
|
|
172
|
+
/** Tall format: the column holding each row's measured value. */
|
|
173
|
+
testvalueCol?: string | null;
|
|
174
|
+
loLimitCol?: string | null;
|
|
175
|
+
hiLimitCol?: string | null;
|
|
176
|
+
unitsCol?: string | null;
|
|
177
|
+
/** Bins counted as a pass in this file's own pass/fail summary. */
|
|
178
|
+
passBins: number[];
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** Every way a parse can fail. Branch on this, never on the message. */
|
|
182
|
+
export type ParseErrorCode =
|
|
183
|
+
| "file-read"
|
|
184
|
+
| "gzip-invalid"
|
|
185
|
+
| "encoding-invalid"
|
|
186
|
+
| "not-stdf"
|
|
187
|
+
| "stdf-unsupported"
|
|
188
|
+
| "csv-read"
|
|
189
|
+
| "json-invalid"
|
|
190
|
+
| "parquet-read"
|
|
191
|
+
| "column-missing"
|
|
192
|
+
| "mapping-invalid"
|
|
193
|
+
/** A worker panicked or a task failed to join — always a bug, never bad input. */
|
|
194
|
+
| "internal";
|
|
195
|
+
|
|
196
|
+
/** What every export below throws. A real `Error`, so `err.message` works and
|
|
197
|
+
* `err instanceof Error` is true, carrying a `code` to branch on. */
|
|
198
|
+
export interface ParserError extends Error {
|
|
199
|
+
name: "ParserError";
|
|
200
|
+
code: ParseErrorCode;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
export function atdf_file_meta(bytes: Uint8Array): FileMeta;
|
|
5
206
|
|
|
6
|
-
export function atdf_test_names(bytes: Uint8Array):
|
|
207
|
+
export function atdf_test_names(bytes: Uint8Array): ScanResult;
|
|
7
208
|
|
|
8
209
|
export function init(): void;
|
|
9
210
|
|
|
10
211
|
/**
|
|
11
212
|
* `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
|
|
12
213
|
*/
|
|
13
|
-
export function parquet_distinct_count(bytes: Uint8Array, columns:
|
|
214
|
+
export function parquet_distinct_count(bytes: Uint8Array, columns: string[]): number;
|
|
14
215
|
|
|
15
|
-
export function parquet_headers(bytes: Uint8Array):
|
|
216
|
+
export function parquet_headers(bytes: Uint8Array): ParquetHeadersResult;
|
|
16
217
|
|
|
17
|
-
export function parse_atdf(bytes: Uint8Array):
|
|
218
|
+
export function parse_atdf(bytes: Uint8Array): ParsedStdf;
|
|
18
219
|
|
|
19
|
-
export function parse_atdf_filtered(bytes: Uint8Array, selected:
|
|
220
|
+
export function parse_atdf_filtered(bytes: Uint8Array, selected: number[]): ParsedStdf;
|
|
20
221
|
|
|
21
|
-
export function parse_csv(bytes: Uint8Array, mapping:
|
|
222
|
+
export function parse_csv(bytes: Uint8Array, mapping: CsvMapping): ParsedStdf;
|
|
22
223
|
|
|
23
|
-
export function parse_json(bytes: Uint8Array, mapping:
|
|
224
|
+
export function parse_json(bytes: Uint8Array, mapping: CsvMapping): ParsedStdf;
|
|
24
225
|
|
|
25
|
-
export function parse_parquet(bytes: Uint8Array, mapping:
|
|
226
|
+
export function parse_parquet(bytes: Uint8Array, mapping: CsvMapping): ParsedStdf;
|
|
26
227
|
|
|
27
|
-
export function parse_stdf(bytes: Uint8Array):
|
|
228
|
+
export function parse_stdf(bytes: Uint8Array): ParsedStdf;
|
|
28
229
|
|
|
29
|
-
export function parse_stdf_filtered(bytes: Uint8Array, selected:
|
|
230
|
+
export function parse_stdf_filtered(bytes: Uint8Array, selected: number[]): ParsedStdf;
|
|
30
231
|
|
|
31
|
-
export function stdf_file_meta(bytes: Uint8Array):
|
|
232
|
+
export function stdf_file_meta(bytes: Uint8Array): FileMeta;
|
|
32
233
|
|
|
33
|
-
export function stdf_test_names(bytes: Uint8Array):
|
|
234
|
+
export function stdf_test_names(bytes: Uint8Array): ScanResult;
|
|
34
235
|
|
|
35
236
|
export type InitInput = RequestInfo | URL | Response | BufferSource | WebAssembly.Module;
|
|
36
237
|
|
package/testdata_parser.js
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* @param {Uint8Array} bytes
|
|
5
|
-
* @returns {
|
|
5
|
+
* @returns {FileMeta}
|
|
6
6
|
*/
|
|
7
7
|
export function atdf_file_meta(bytes) {
|
|
8
8
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -16,7 +16,7 @@ export function atdf_file_meta(bytes) {
|
|
|
16
16
|
|
|
17
17
|
/**
|
|
18
18
|
* @param {Uint8Array} bytes
|
|
19
|
-
* @returns {
|
|
19
|
+
* @returns {ScanResult}
|
|
20
20
|
*/
|
|
21
21
|
export function atdf_test_names(bytes) {
|
|
22
22
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -35,7 +35,7 @@ export function init() {
|
|
|
35
35
|
/**
|
|
36
36
|
* `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
|
|
37
37
|
* @param {Uint8Array} bytes
|
|
38
|
-
* @param {
|
|
38
|
+
* @param {string[]} columns
|
|
39
39
|
* @returns {number}
|
|
40
40
|
*/
|
|
41
41
|
export function parquet_distinct_count(bytes, columns) {
|
|
@@ -50,7 +50,7 @@ export function parquet_distinct_count(bytes, columns) {
|
|
|
50
50
|
|
|
51
51
|
/**
|
|
52
52
|
* @param {Uint8Array} bytes
|
|
53
|
-
* @returns {
|
|
53
|
+
* @returns {ParquetHeadersResult}
|
|
54
54
|
*/
|
|
55
55
|
export function parquet_headers(bytes) {
|
|
56
56
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -64,7 +64,7 @@ export function parquet_headers(bytes) {
|
|
|
64
64
|
|
|
65
65
|
/**
|
|
66
66
|
* @param {Uint8Array} bytes
|
|
67
|
-
* @returns {
|
|
67
|
+
* @returns {ParsedStdf}
|
|
68
68
|
*/
|
|
69
69
|
export function parse_atdf(bytes) {
|
|
70
70
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -78,8 +78,8 @@ export function parse_atdf(bytes) {
|
|
|
78
78
|
|
|
79
79
|
/**
|
|
80
80
|
* @param {Uint8Array} bytes
|
|
81
|
-
* @param {
|
|
82
|
-
* @returns {
|
|
81
|
+
* @param {number[]} selected
|
|
82
|
+
* @returns {ParsedStdf}
|
|
83
83
|
*/
|
|
84
84
|
export function parse_atdf_filtered(bytes, selected) {
|
|
85
85
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -93,8 +93,8 @@ export function parse_atdf_filtered(bytes, selected) {
|
|
|
93
93
|
|
|
94
94
|
/**
|
|
95
95
|
* @param {Uint8Array} bytes
|
|
96
|
-
* @param {
|
|
97
|
-
* @returns {
|
|
96
|
+
* @param {CsvMapping} mapping
|
|
97
|
+
* @returns {ParsedStdf}
|
|
98
98
|
*/
|
|
99
99
|
export function parse_csv(bytes, mapping) {
|
|
100
100
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -108,8 +108,8 @@ export function parse_csv(bytes, mapping) {
|
|
|
108
108
|
|
|
109
109
|
/**
|
|
110
110
|
* @param {Uint8Array} bytes
|
|
111
|
-
* @param {
|
|
112
|
-
* @returns {
|
|
111
|
+
* @param {CsvMapping} mapping
|
|
112
|
+
* @returns {ParsedStdf}
|
|
113
113
|
*/
|
|
114
114
|
export function parse_json(bytes, mapping) {
|
|
115
115
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -123,8 +123,8 @@ export function parse_json(bytes, mapping) {
|
|
|
123
123
|
|
|
124
124
|
/**
|
|
125
125
|
* @param {Uint8Array} bytes
|
|
126
|
-
* @param {
|
|
127
|
-
* @returns {
|
|
126
|
+
* @param {CsvMapping} mapping
|
|
127
|
+
* @returns {ParsedStdf}
|
|
128
128
|
*/
|
|
129
129
|
export function parse_parquet(bytes, mapping) {
|
|
130
130
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -138,7 +138,7 @@ export function parse_parquet(bytes, mapping) {
|
|
|
138
138
|
|
|
139
139
|
/**
|
|
140
140
|
* @param {Uint8Array} bytes
|
|
141
|
-
* @returns {
|
|
141
|
+
* @returns {ParsedStdf}
|
|
142
142
|
*/
|
|
143
143
|
export function parse_stdf(bytes) {
|
|
144
144
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -152,8 +152,8 @@ export function parse_stdf(bytes) {
|
|
|
152
152
|
|
|
153
153
|
/**
|
|
154
154
|
* @param {Uint8Array} bytes
|
|
155
|
-
* @param {
|
|
156
|
-
* @returns {
|
|
155
|
+
* @param {number[]} selected
|
|
156
|
+
* @returns {ParsedStdf}
|
|
157
157
|
*/
|
|
158
158
|
export function parse_stdf_filtered(bytes, selected) {
|
|
159
159
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -167,7 +167,7 @@ export function parse_stdf_filtered(bytes, selected) {
|
|
|
167
167
|
|
|
168
168
|
/**
|
|
169
169
|
* @param {Uint8Array} bytes
|
|
170
|
-
* @returns {
|
|
170
|
+
* @returns {FileMeta}
|
|
171
171
|
*/
|
|
172
172
|
export function stdf_file_meta(bytes) {
|
|
173
173
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -181,7 +181,7 @@ export function stdf_file_meta(bytes) {
|
|
|
181
181
|
|
|
182
182
|
/**
|
|
183
183
|
* @param {Uint8Array} bytes
|
|
184
|
-
* @returns {
|
|
184
|
+
* @returns {ScanResult}
|
|
185
185
|
*/
|
|
186
186
|
export function stdf_test_names(bytes) {
|
|
187
187
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
@@ -327,6 +327,10 @@ function __wbg_get_imports() {
|
|
|
327
327
|
const ret = new Error();
|
|
328
328
|
return ret;
|
|
329
329
|
},
|
|
330
|
+
__wbg_new_50bb5ebeecef71a8: function(arg0, arg1) {
|
|
331
|
+
const ret = new Error(getStringFromWasm0(arg0, arg1));
|
|
332
|
+
return ret;
|
|
333
|
+
},
|
|
330
334
|
__wbg_new_578aeef4b6b94378: function(arg0) {
|
|
331
335
|
const ret = new Uint8Array(arg0);
|
|
332
336
|
return ret;
|
|
@@ -361,9 +365,16 @@ function __wbg_get_imports() {
|
|
|
361
365
|
__wbg_set_6be42768c690e380: function(arg0, arg1, arg2) {
|
|
362
366
|
arg0[arg1] = arg2;
|
|
363
367
|
},
|
|
368
|
+
__wbg_set_6e30c9374c26414c: function() { return handleError(function (arg0, arg1, arg2) {
|
|
369
|
+
const ret = Reflect.set(arg0, arg1, arg2);
|
|
370
|
+
return ret;
|
|
371
|
+
}, arguments); },
|
|
364
372
|
__wbg_set_dca99999bba88a9a: function(arg0, arg1, arg2) {
|
|
365
373
|
arg0[arg1 >>> 0] = arg2;
|
|
366
374
|
},
|
|
375
|
+
__wbg_set_name_3c3fc49c8f747e56: function(arg0, arg1, arg2) {
|
|
376
|
+
arg0.name = getStringFromWasm0(arg1, arg2);
|
|
377
|
+
},
|
|
367
378
|
__wbg_stack_3b0d974bbf31e44f: function(arg0, arg1) {
|
|
368
379
|
const ret = arg1.stack;
|
|
369
380
|
const ptr1 = passStringToWasm0(ret, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
package/testdata_parser_bg.wasm
CHANGED
|
Binary file
|