@wafertools/testdata-parser 0.14.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/llms.txt +2 -1
- package/package.json +1 -1
- package/testdata_parser.d.ts +2 -2
- package/testdata_parser.js +3 -2
- package/testdata_parser_bg.wasm +0 -0
package/README.md
CHANGED
|
@@ -84,7 +84,7 @@ Every parse function takes raw file bytes (`Uint8Array`), or throws a `ParserErr
|
|
|
84
84
|
| `atdf_test_names` | `(bytes: Uint8Array) => ScanResult` | Same first-pass scan for ATDF |
|
|
85
85
|
| `stdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Lot metadata, wafer count, first/last timestamps and site count from an MIR/SDR/WIR/WRR-only scan — no PTR/FTR/PIR/PRR walk, so it stays cheap across a batch of files |
|
|
86
86
|
| `atdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Same metadata-only scan for ATDF |
|
|
87
|
-
| `parquet_distinct_count` | `(bytes: Uint8Array, columns: string[]) => number` | How many distinct combinations of those columns the file holds — a wafer count from `['lot','wafer']` without a full parse, read as a column projection. A column missing from the schema is an error, not a count of zero. Parquet only: CSV/JSON have no equivalent shortcut |
|
|
87
|
+
| `parquet_distinct_count` | `(bytes: Uint8Array, columns: string[], blankIsAValue: boolean) => number` | How many distinct combinations of those columns the file holds — a wafer count from `['lot','wafer']` without a full parse, read as a column projection. `blankIsAValue` is stated by every caller: `false` skips rows blank in every named column (a wafer count: an empty row is not a wafer), `true` counts them as one more value (is this column constant? a column that is "25" in some rows and empty in the rest is not). A column missing from the schema is an error, not a count of zero. Parquet only: CSV/JSON have no equivalent shortcut |
|
|
88
88
|
| `parse_stdf_filtered` | `(bytes: Uint8Array, selected: number[]) => Uint8Array` | Full parse, skipping per-site accumulation for test numbers not in `selected` |
|
|
89
89
|
| `parse_atdf_filtered` | `(bytes: Uint8Array, selected: number[]) => Uint8Array` | Same filtered parse for ATDF |
|
|
90
90
|
|
|
@@ -336,7 +336,7 @@ There is no `csv_headers`/`json_headers` in the WASM API — a browser caller th
|
|
|
336
336
|
```ts
|
|
337
337
|
interface ParquetHeadersResult {
|
|
338
338
|
headers: string[];
|
|
339
|
-
sample: Record<string, string>[];
|
|
339
|
+
sample: Record<string, string>[]; // the first five rows, then the first row of up to twenty later row groups, evenly spaced
|
|
340
340
|
rowCount: number;
|
|
341
341
|
columnTypes: Record<string, 'number' | 'bool' | 'string'>;
|
|
342
342
|
}
|
|
@@ -358,7 +358,7 @@ The crate also builds as a native Rust library (used directly by tsmap's Tauri c
|
|
|
358
358
|
| `parse_json_sync(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
|
|
359
359
|
| `parquet_headers_inner(path: String) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
|
|
360
360
|
| `parse_parquet_inner(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
|
|
361
|
-
| `parquet_distinct_count_inner(path: String, columns: Vec<String
|
|
361
|
+
| `parquet_distinct_count_inner(path: String, columns: Vec<String>, blank_is_a_value: bool) -> ParseResult<usize>` | `parse_parquet` |
|
|
362
362
|
| `read_bytes(path: &str) -> ParseResult<Vec<u8>>` | `read_file` |
|
|
363
363
|
| `read_text(path: &str) -> ParseResult<String>` | `read_file` |
|
|
364
364
|
|
|
@@ -383,7 +383,7 @@ alone. Same codes as the WASM layer above; it is the same error, serialised ther
|
|
|
383
383
|
| `parse_json_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
|
|
384
384
|
| `parquet_headers_from_bytes(&[u8]) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
|
|
385
385
|
| `parse_parquet_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
|
|
386
|
-
| `parquet_distinct_count_from_bytes(&[u8], &[String]) -> ParseResult<usize>` | `parse_parquet` |
|
|
386
|
+
| `parquet_distinct_count_from_bytes(&[u8], &[String], bool) -> ParseResult<usize>` | `parse_parquet` |
|
|
387
387
|
| `decompress_if_gzip(Vec<u8>) -> ParseResult<Vec<u8>>` | `read_file` |
|
|
388
388
|
|
|
389
389
|
`CsvHeadersResult` and `JsonHeadersResult` are the same shape — the header row plus enough of the file to preview a mapping:
|
|
@@ -391,7 +391,7 @@ alone. Same codes as the WASM layer above; it is the same error, serialised ther
|
|
|
391
391
|
```rust
|
|
392
392
|
pub struct CsvHeadersResult {
|
|
393
393
|
pub headers: Vec<String>,
|
|
394
|
-
pub sample: Vec<HashMap<String, String>>, // first
|
|
394
|
+
pub sample: Vec<HashMap<String, String>>, // the first five rows, then up to twenty spread through the file
|
|
395
395
|
pub row_count: usize,
|
|
396
396
|
}
|
|
397
397
|
```
|
package/llms.txt
CHANGED
|
@@ -31,7 +31,8 @@ resolve.
|
|
|
31
31
|
- `parse_csv(bytes, mapping)` / `parse_json(bytes, mapping)` / `parse_parquet(bytes, mapping)` - full parse with an explicit `CsvMapping`. There is no header auto-detection.
|
|
32
32
|
- `stdf_test_names(bytes)` / `atdf_test_names(bytes)` - fast scan: test definitions + die count, no per-die accumulation.
|
|
33
33
|
- `parse_stdf_filtered(bytes, selected)` / `parse_atdf_filtered(bytes, selected)` - full parse that accumulates values only for the selected test numbers.
|
|
34
|
-
- `parquet_headers(bytes)` - schema, sample rows
|
|
34
|
+
- `parquet_headers(bytes)` - schema, sample rows (the first five, then the first row of up to twenty later row groups) and per-column types, for a mapping UI.
|
|
35
|
+
- `parquet_distinct_count(bytes, columns, blankIsAValue)` - distinct combinations of the named columns, read as a projection. `blankIsAValue: false` skips rows blank in every named column (a wafer count); `true` counts them as a value (is this column constant?). There is deliberately no `csv_headers`/`json_headers` in the WASM API: read a text header row in plain JS.
|
|
35
36
|
- `stdf_file_meta(bytes)` / `atdf_file_meta(bytes)` - lot metadata, wafer count, first/last timestamps and site count, without a full parse.
|
|
36
37
|
|
|
37
38
|
## Traps
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "@wafertools/testdata-parser",
|
|
3
3
|
"type": "module",
|
|
4
4
|
"description": "Rust/WASM parsers for semiconductor test data formats (STDF, ATDF, CSV, JSON, Parquet)",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.15.0",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"repository": {
|
|
8
8
|
"type": "git",
|
package/testdata_parser.d.ts
CHANGED
|
@@ -240,7 +240,7 @@ export function init(): void;
|
|
|
240
240
|
/**
|
|
241
241
|
* `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
|
|
242
242
|
*/
|
|
243
|
-
export function parquet_distinct_count(bytes: Uint8Array, columns: string[]): number;
|
|
243
|
+
export function parquet_distinct_count(bytes: Uint8Array, columns: string[], blank_is_a_value: boolean): number;
|
|
244
244
|
|
|
245
245
|
export function parquet_headers(bytes: Uint8Array): ParquetHeadersResult;
|
|
246
246
|
|
|
@@ -269,7 +269,7 @@ export interface InitOutput {
|
|
|
269
269
|
readonly atdf_file_meta: (a: number, b: number) => [number, number, number];
|
|
270
270
|
readonly atdf_test_names: (a: number, b: number) => [number, number, number];
|
|
271
271
|
readonly init: () => void;
|
|
272
|
-
readonly parquet_distinct_count: (a: number, b: number, c: any) => [number, number, number];
|
|
272
|
+
readonly parquet_distinct_count: (a: number, b: number, c: any, d: number) => [number, number, number];
|
|
273
273
|
readonly parquet_headers: (a: number, b: number) => [number, number, number];
|
|
274
274
|
readonly parse_atdf: (a: number, b: number) => [number, number, number, number];
|
|
275
275
|
readonly parse_atdf_filtered: (a: number, b: number, c: any) => [number, number, number, number];
|
package/testdata_parser.js
CHANGED
|
@@ -36,12 +36,13 @@ export function init() {
|
|
|
36
36
|
* `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
|
|
37
37
|
* @param {Uint8Array} bytes
|
|
38
38
|
* @param {string[]} columns
|
|
39
|
+
* @param {boolean} blank_is_a_value
|
|
39
40
|
* @returns {number}
|
|
40
41
|
*/
|
|
41
|
-
export function parquet_distinct_count(bytes, columns) {
|
|
42
|
+
export function parquet_distinct_count(bytes, columns, blank_is_a_value) {
|
|
42
43
|
const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
|
|
43
44
|
const len0 = WASM_VECTOR_LEN;
|
|
44
|
-
const ret = wasm.parquet_distinct_count(ptr0, len0, columns);
|
|
45
|
+
const ret = wasm.parquet_distinct_count(ptr0, len0, columns, blank_is_a_value);
|
|
45
46
|
if (ret[2]) {
|
|
46
47
|
throw takeFromExternrefTable0(ret[1]);
|
|
47
48
|
}
|
package/testdata_parser_bg.wasm
CHANGED
|
Binary file
|