@wafertools/testdata-parser 0.11.1 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -7
- package/llms.txt +1 -1
- package/package.json +1 -1
- package/testdata_parser.d.ts +30 -2
- package/testdata_parser_bg.wasm +0 -0
package/README.md
CHANGED
|
@@ -116,8 +116,10 @@ interface CsvMapping {
|
|
|
116
116
|
testnameCol?: string; // for "tall" CSVs: column holding the test name per row
|
|
117
117
|
testnumberCol?: string; // for "tall" CSVs: column holding the test's real number per row
|
|
118
118
|
testvalueCol?: string; // for "tall" CSVs: column holding the test value per row
|
|
119
|
-
loLimitCol?: string;
|
|
119
|
+
loLimitCol?: string; // for "tall" CSVs: test limits (STDF LO_LIMIT/HI_LIMIT) per row
|
|
120
120
|
hiLimitCol?: string;
|
|
121
|
+
loSpecCol?: string; // for "tall" CSVs: spec limits (STDF LO_SPEC/HI_SPEC) per row — a separate pair
|
|
122
|
+
hiSpecCol?: string;
|
|
121
123
|
unitsCol?: string;
|
|
122
124
|
passBins: number[]; // hbin/sbin values treated as a pass for pass/fail summary
|
|
123
125
|
}
|
|
@@ -157,6 +159,7 @@ interface ParsedStdf {
|
|
|
157
159
|
|
|
158
160
|
interface WaferData {
|
|
159
161
|
waferId: string;
|
|
162
|
+
waferIdPlaceholder?: boolean; // true when the file gave the wafer no ID (waferId is W1, W2…)
|
|
160
163
|
results: DieResult[];
|
|
161
164
|
partCount?: number;
|
|
162
165
|
goodCount?: number;
|
|
@@ -171,7 +174,8 @@ interface DieResult {
|
|
|
171
174
|
hbin?: number;
|
|
172
175
|
sbin?: number;
|
|
173
176
|
siteNum?: number;
|
|
174
|
-
partId?:
|
|
177
|
+
partId?: string; // STDF/ATDF PART_ID as text; data, not an identifier
|
|
178
|
+
supersedes?: 'partId' | 'position'; // the tester marked this record as replacing an earlier one
|
|
175
179
|
testValues?: Record<string, number>; // keyed by test number as a string
|
|
176
180
|
testPass?: Record<string, boolean>; // recorded verdicts, true = pass; see below
|
|
177
181
|
}
|
|
@@ -181,6 +185,10 @@ interface TestDef {
|
|
|
181
185
|
testType: string; // "P" (parametric) or "F" (functional)
|
|
182
186
|
loLimit?: number;
|
|
183
187
|
hiLimit?: number;
|
|
188
|
+
loSpec?: number; // specification limits (STDF LO_SPEC/HI_SPEC), separate from the test limits
|
|
189
|
+
hiSpec?: number;
|
|
190
|
+
loLimitInclusive?: boolean; // false: a result equal to the low test limit fails (STDF PARM_FLG bit 6 clear, ATDF "L")
|
|
191
|
+
hiLimitInclusive?: boolean; // the same for the high test limit (PARM_FLG bit 7, ATDF "H"); absent = equal passes
|
|
184
192
|
units?: string;
|
|
185
193
|
order?: number; // display order, independent of the key — see below
|
|
186
194
|
}
|
|
@@ -212,7 +220,14 @@ interface ParserWarning {
|
|
|
212
220
|
|
|
213
221
|
type ParserWarningCode =
|
|
214
222
|
| 'unpositioned-dies' // dies with no X/Y — real data, not placeable
|
|
215
|
-
| '
|
|
223
|
+
| 'bin-invalid' // a bin outside STDF's 0–32767, or a missing hard bin
|
|
224
|
+
| 'coordinate-invalid' // an X/Y outside STDF's -32767..32767
|
|
225
|
+
| 'site-invalid' // a site number outside STDF's 0–255 (flat formats)
|
|
226
|
+
| 'result-unusable' // results the tester flagged unusable (value left out, verdict kept)
|
|
227
|
+
| 'records-not-read' // records this parser does not read yet (MPR)
|
|
228
|
+
| 'wafer-end-missing' // a wafer had no WRR; closed at the next wafer or end of file
|
|
229
|
+
| 'file-truncated' // the file ends part-way through a record
|
|
230
|
+
| 'record-malformed' // a PRR too short to hold its required fields
|
|
216
231
|
| 'values-not-numeric' // a mapped column held values that would not coerce
|
|
217
232
|
| 'retests-assumed' // repeated positions read as retests
|
|
218
233
|
| 'wafer-split-by-column' // one wafer per value of a mapped column
|
|
@@ -250,9 +265,10 @@ an empty array, which asserts that nothing passes.
|
|
|
250
265
|
**`warnings` carries a stable `code`, prose, and a severity** — branch on the code, display
|
|
251
266
|
the message, and never match on the prose. `severity: 'error'` means a number or a plot
|
|
252
267
|
built from this result can mislead, because data was dropped or a value was substituted
|
|
253
|
-
(`unpositioned-dies`, `
|
|
254
|
-
parse
|
|
255
|
-
|
|
268
|
+
(`unpositioned-dies`, `bin-invalid`, `coordinate-invalid`, `site-invalid`, `record-malformed`, `records-not-read`, `values-not-numeric`); `'warning'` means the
|
|
269
|
+
parse applied a documented rule or interpretation — one you may want to change, or, like
|
|
270
|
+
`result-unusable`, the spec's own rule for leaving out values the tester flagged — and the
|
|
271
|
+
result means what the file says. Nothing here is fatal — the parse succeeded. Surface them: a silently discarded
|
|
256
272
|
warning is how a partly-wrong load looks fine.
|
|
257
273
|
|
|
258
274
|
`testDefs` and `testValues` are both keyed by **test number**, not test name — test numbers are the unique identity in STDF/ATDF; names are not guaranteed unique.
|
|
@@ -362,7 +378,16 @@ pub struct CsvHeadersResult {
|
|
|
362
378
|
- **Byte readers are panic-free.** STDF/ATDF field readers are bounds-checked and return `Option`/`Result` rather than panicking on truncated input — a panic inside WASM aborts the whole module with no recovery, so this is a hard requirement, not a style preference.
|
|
363
379
|
- **Big-endian and little-endian STDF** are both supported (detected from the FAR record's `CPU_TYPE`).
|
|
364
380
|
- **Gzip is transparent** — every entry point decompresses `.gz` input automatically by sniffing the magic bytes; callers don't need to branch on compression.
|
|
365
|
-
- **
|
|
381
|
+
- **MPR records are not read yet.** A multiple-result parametric record (STDF `MPR`, ATDF `MPR:`)
|
|
382
|
+
carries several results for one test; both parsers skip them — in the full parse and the
|
|
383
|
+
test-name scan alike — and count them in a `records-not-read` warning, so the missing tests are
|
|
384
|
+
never silent. PTR and FTR records are read in full.
|
|
385
|
+
- **Every format applies STDF V4's value ranges.** A bin, coordinate or site number STDF cannot
|
|
386
|
+
store, or a test value that is not finite, is missing in CSV, JSON and Parquet exactly as in
|
|
387
|
+
STDF and ATDF, and is reported under the same codes (`bin-invalid`, `coordinate-invalid`,
|
|
388
|
+
`site-invalid`, `result-unusable`). The rule lives in one place (`SpecCheck` in `types.rs`),
|
|
389
|
+
which the flat formats reach through `flat_wafers::into_parsed`.
|
|
390
|
+
- **CSV/JSON/Parquet test numbers fall back to a deterministic hash only when the file itself carries no real one.** Neither format has a *mandatory* STDF-style test number the way STDF/ATDF do, but a real one is used whenever the source data has it — see "Test identity — real number vs. synthesized one" above for the wide/tall rules. Hashing is the fallback, not the default: it fires per test only when no real number was mapped or the mapped column's value didn't parse (`test_identity::stable_test_number`, FNV-1a with a fixed seed and a reserved floor, collision-probed so two tests in one file can never collide, and never colliding with a genuine numeric-header/`testnumberCol` value either). Deliberately not sequential/encounter-order: a hash means the number for a given test doesn't change if the file is reordered or a column is added — a *hashed* number is otherwise meaningless and callers should never rely on its value, only on it being stable and unique within one parse. `order` (see `TestDef` above) carries the file's own display order instead.
|
|
366
391
|
- **Parquet reads through a row-oriented API, not Arrow.** `parquet::record::Row`/`Field` rather than the `arrow` feature — a closer fit for this crate's row-based `DieResult` model, and a smaller WASM bundle (no Arrow array machinery pulled in). A typed Parquet cell is coerced to `f64` for numeric roles and to a plain string otherwise; a value that fails to coerce (e.g. a numeric role mapped to a genuinely string-typed column) is skipped and surfaced as one summarised entry in `warnings`, not a panic or a silent zero.
|
|
367
392
|
- **Parquet's `zstd` codec is native-only.** `snappy`, `gzip`, `lz4`, and `brotli` build for `wasm32-unknown-unknown` with no extra toolchain; `zstd`'s C library needs a real C cross-compiler targeting wasm32, which a plain `wasm-pack build` doesn't assume is available. A `zstd`-compressed Parquet file parses natively but fails clearly on the WASM build.
|
|
368
393
|
|
package/llms.txt
CHANGED
|
@@ -37,7 +37,7 @@ resolve.
|
|
|
37
37
|
- **Failures throw a `ParserError` — branch on `err.code`, display `err.message`.** It is a real `Error` (so `err.message` reads and `err instanceof Error` is true) carrying a stable code: `not-stdf`, `stdf-unsupported`, `file-read`, `gzip-invalid`, `encoding-invalid`, `csv-read`, `json-invalid`, `parquet-read`, `column-missing`, `mapping-invalid`, `internal`. Never match on the message — messages are prose and get reworded. `ParseErrorCode` is a union in the declarations, so a `switch` is exhaustively checked.
|
|
38
38
|
- **The package ships real TypeScript declarations — use them.** `ParsedStdf`, `DieResult`, `TestDef`, `ParserWarning`, `ScanResult`, `FileMeta`, `ParquetHeadersResult`, `CsvMapping` and the two code unions are all in `testdata_parser.d.ts`, and every export is typed with its real return type. Do not redeclare these shapes in your own code, and do not guess field names: let the compiler tell you. (Versions before this one typed every return value as `any` — if that is what you see, the installed version is older than the declarations.)
|
|
39
39
|
- **`x` and `y` are optional.** A die with no reported position has neither (never one of the two) and carries `dieIndex` instead: it still holds real measured data and counts toward non-spatial statistics, but cannot be placed on a wafer map. Filter those out before building a map; do not default them to 0, which invents a die at the origin.
|
|
40
|
-
- **CSV/JSON/Parquet
|
|
40
|
+
- **CSV/JSON/Parquet only hash a test number when the source has no real one.** Neither format has a *mandatory* test number the way STDF/ATDF do, but if the source carries real numbers, use them: `testnumberCol` (tall layout) or the column's own numeric header (wide layout) are used as-is, never hashed. Hashing is strictly the fallback — in a tall layout the crate hashes the test's identity (FNV-1a, fixed seed, collision-probed) only when no real number was mapped or it failed to parse; in a wide layout the caller assigns `CsvTestCol.testNumber` when building the mapping, same rule. A *hashed* number is stable and unique *within one parse*, and that is all — never display it to a user, persist it, or join two files on it. Hashed numbers are always >= 1,000,000, a band chosen so one can never be mistaken for a real STDF test number, a 1001-style sequential one, or a genuine numeric-header/`testnumberCol` value. `TestDef.order` carries the file's own display order; STDF/ATDF omit it, because there the real test number already sorts meaningfully.
|
|
41
41
|
- **Two-pass parsing loses test metadata unless you merge it back.** `testDefs` in a filtered result only covers tests seen in the second pass, so a test that appeared only on an early stop-on-fail die can vanish. Merge the first-pass `ScanResult.testDefs` into the filtered result.
|
|
42
42
|
- **Empty collections are omitted, not emitted.** `warnings`, `hbinDefs`, `sbinDefs`, `passHbins`, `testValues` and `testPass` are all absent from the object when empty, so read them with `?.` and treat missing as empty.
|
|
43
43
|
- **Warnings are structured, and half of them mean a number could be wrong.** Each is `{ code, message, severity }`. `severity: 'error'` means data was dropped or a value substituted, so a plot or yield built from the parse can mislead — `unpositioned-dies`, `soft-bin-mirrored`, `values-not-numeric`. `'warning'` means the parse made a documented interpretation you may want to change — `retests-assumed`, `wafer-split-by-column`, `column-varies-within-wafer`, `multiple-lot-records`. Branch on `code`, never the prose, and surface them: a discarded advisory is how a partly-wrong load looks fine.
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "@wafertools/testdata-parser",
|
|
3
3
|
"type": "module",
|
|
4
4
|
"description": "Rust/WASM parsers for semiconductor test data formats (STDF, ATDF, CSV, JSON, Parquet)",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.13.0",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"repository": {
|
|
8
8
|
"type": "git",
|
package/testdata_parser.d.ts
CHANGED
|
@@ -27,9 +27,19 @@ export interface TestDef {
|
|
|
27
27
|
name: string;
|
|
28
28
|
/** "P" parametric (has a measured value) or "F" functional (verdict only). */
|
|
29
29
|
testType: "P" | "F";
|
|
30
|
+
/** Test limits — the pass/fail limits (STDF LO_LIMIT/HI_LIMIT). */
|
|
30
31
|
loLimit?: number;
|
|
31
32
|
hiLimit?: number;
|
|
32
33
|
units?: string;
|
|
34
|
+
/** Specification limits (STDF LO_SPEC/HI_SPEC) — what process capability is
|
|
35
|
+
* judged against. Separate from the test limits. */
|
|
36
|
+
loSpec?: number;
|
|
37
|
+
hiSpec?: number;
|
|
38
|
+
/** `false` when a result equal to the low test limit fails (STDF PARM_FLG
|
|
39
|
+
* bit 6 clear, ATDF Limit Compare `L`). Absent = it passes. */
|
|
40
|
+
loLimitInclusive?: boolean;
|
|
41
|
+
/** The same for the high test limit (STDF PARM_FLG bit 7, ATDF `H`). */
|
|
42
|
+
hiLimitInclusive?: boolean;
|
|
33
43
|
/** The file's own display order. Absent for STDF/ATDF, where the real test
|
|
34
44
|
* number already sorts meaningfully. */
|
|
35
45
|
order?: number;
|
|
@@ -47,7 +57,11 @@ export interface DieResult {
|
|
|
47
57
|
hbin?: number;
|
|
48
58
|
sbin?: number;
|
|
49
59
|
siteNum?: number;
|
|
50
|
-
|
|
60
|
+
/** STDF/ATDF PART_ID, as text. Data about the part, not an identifier. */
|
|
61
|
+
partId?: string;
|
|
62
|
+
/** The tester marked this record as replacing an earlier one: with the same
|
|
63
|
+
* part ID (`"partId"`) or at the same X/Y (`"position"`). */
|
|
64
|
+
supersedes?: "partId" | "position";
|
|
51
65
|
/** Measured values, keyed by test number as a string. Parametric tests only. */
|
|
52
66
|
testValues?: Record<string, number>;
|
|
53
67
|
/** Recorded pass/fail verdicts, `true` = pass, keyed like `testValues`.
|
|
@@ -58,6 +72,8 @@ export interface DieResult {
|
|
|
58
72
|
|
|
59
73
|
export interface WaferData {
|
|
60
74
|
waferId: string;
|
|
75
|
+
/** `waferId` is a placeholder (`W1`, `W2`…): the file gave this wafer no ID. */
|
|
76
|
+
waferIdPlaceholder?: boolean;
|
|
61
77
|
results: DieResult[];
|
|
62
78
|
partCount?: number;
|
|
63
79
|
goodCount?: number;
|
|
@@ -69,7 +85,14 @@ export interface WaferData {
|
|
|
69
85
|
/** Every advisory a parse can raise. Branch on this, never on the message. */
|
|
70
86
|
export type ParserWarningCode =
|
|
71
87
|
| "unpositioned-dies"
|
|
72
|
-
| "
|
|
88
|
+
| "bin-invalid"
|
|
89
|
+
| "coordinate-invalid"
|
|
90
|
+
| "site-invalid"
|
|
91
|
+
| "result-unusable"
|
|
92
|
+
| "records-not-read"
|
|
93
|
+
| "wafer-end-missing"
|
|
94
|
+
| "file-truncated"
|
|
95
|
+
| "record-malformed"
|
|
73
96
|
| "values-not-numeric"
|
|
74
97
|
| "retests-assumed"
|
|
75
98
|
| "wafer-split-by-column"
|
|
@@ -171,8 +194,13 @@ export interface CsvMapping {
|
|
|
171
194
|
testnumberCol?: string | null;
|
|
172
195
|
/** Tall format: the column holding each row's measured value. */
|
|
173
196
|
testvalueCol?: string | null;
|
|
197
|
+
/** Tall format: the columns holding each test's test limits (STDF LO_LIMIT/HI_LIMIT). */
|
|
174
198
|
loLimitCol?: string | null;
|
|
175
199
|
hiLimitCol?: string | null;
|
|
200
|
+
/** Tall format: the columns holding each test's spec limits (STDF LO_SPEC/HI_SPEC) —
|
|
201
|
+
* a separate pair from the test limits, never mixed with them. */
|
|
202
|
+
loSpecCol?: string | null;
|
|
203
|
+
hiSpecCol?: string | null;
|
|
176
204
|
unitsCol?: string | null;
|
|
177
205
|
/** Bins counted as a pass in this file's own pass/fail summary. */
|
|
178
206
|
passBins: number[];
|
package/testdata_parser_bg.wasm
CHANGED
|
Binary file
|