@wafertools/testdata-parser 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -84,7 +84,7 @@ Every parse function takes raw file bytes (`Uint8Array`), or throws a `ParserErr
84
84
  | `atdf_test_names` | `(bytes: Uint8Array) => ScanResult` | Same first-pass scan for ATDF |
85
85
  | `stdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Lot metadata, wafer count, first/last timestamps and site count from an MIR/SDR/WIR/WRR-only scan — no PTR/FTR/PIR/PRR walk, so it stays cheap across a batch of files |
86
86
  | `atdf_file_meta` | `(bytes: Uint8Array) => FileMeta` | Same metadata-only scan for ATDF |
87
- | `parquet_distinct_count` | `(bytes: Uint8Array, columns: string[]) => number` | How many distinct combinations of those columns the file holds — a wafer count from `['lot','wafer']` without a full parse, read as a column projection. A column missing from the schema is an error, not a count of zero. Parquet only: CSV/JSON have no equivalent shortcut |
87
+ | `parquet_distinct_count` | `(bytes: Uint8Array, columns: string[], blankIsAValue: boolean) => number` | How many distinct combinations of those columns the file holds — a wafer count from `['lot','wafer']` without a full parse, read as a column projection. `blankIsAValue` is stated by every caller: `false` skips rows blank in every named column (a wafer count: an empty row is not a wafer), `true` counts them as one more value (is this column constant? a column that is "25" in some rows and empty in the rest is not). A column missing from the schema is an error, not a count of zero. Parquet only: CSV/JSON have no equivalent shortcut |
88
88
  | `parse_stdf_filtered` | `(bytes: Uint8Array, selected: number[]) => Uint8Array` | Full parse, skipping per-site accumulation for test numbers not in `selected` |
89
89
  | `parse_atdf_filtered` | `(bytes: Uint8Array, selected: number[]) => Uint8Array` | Same filtered parse for ATDF |
90
90
 
@@ -336,7 +336,7 @@ There is no `csv_headers`/`json_headers` in the WASM API — a browser caller th
336
336
  ```ts
337
337
  interface ParquetHeadersResult {
338
338
  headers: string[];
339
- sample: Record<string, string>[];
339
+ sample: Record<string, string>[]; // the first five rows, then the first row of up to twenty later row groups, evenly spaced
340
340
  rowCount: number;
341
341
  columnTypes: Record<string, 'number' | 'bool' | 'string'>;
342
342
  }
@@ -358,7 +358,7 @@ The crate also builds as a native Rust library (used directly by tsmap's Tauri c
358
358
  | `parse_json_sync(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
359
359
  | `parquet_headers_inner(path: String) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
360
360
  | `parse_parquet_inner(path: String, mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
361
- | `parquet_distinct_count_inner(path: String, columns: Vec<String>) -> ParseResult<usize>` | `parse_parquet` |
361
+ | `parquet_distinct_count_inner(path: String, columns: Vec<String>, blank_is_a_value: bool) -> ParseResult<usize>` | `parse_parquet` |
362
362
  | `read_bytes(path: &str) -> ParseResult<Vec<u8>>` | `read_file` |
363
363
  | `read_text(path: &str) -> ParseResult<String>` | `read_file` |
364
364
 
@@ -383,7 +383,7 @@ alone. Same codes as the WASM layer above; it is the same error, serialised ther
383
383
  | `parse_json_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_json` |
384
384
  | `parquet_headers_from_bytes(&[u8]) -> ParseResult<ParquetHeadersResult>` | `parse_parquet` |
385
385
  | `parse_parquet_from_bytes(&[u8], mapping: CsvMapping) -> ParseResult<ParsedStdf>` | `parse_parquet` |
386
- | `parquet_distinct_count_from_bytes(&[u8], &[String]) -> ParseResult<usize>` | `parse_parquet` |
386
+ | `parquet_distinct_count_from_bytes(&[u8], &[String], bool) -> ParseResult<usize>` | `parse_parquet` |
387
387
  | `decompress_if_gzip(Vec<u8>) -> ParseResult<Vec<u8>>` | `read_file` |
388
388
 
389
389
  `CsvHeadersResult` and `JsonHeadersResult` are the same shape — the header row plus enough of the file to preview a mapping:
@@ -391,7 +391,7 @@ alone. Same codes as the WASM layer above; it is the same error, serialised ther
391
391
  ```rust
392
392
  pub struct CsvHeadersResult {
393
393
  pub headers: Vec<String>,
394
- pub sample: Vec<HashMap<String, String>>, // first few rows, for a preview UI
394
+ pub sample: Vec<HashMap<String, String>>, // the first five rows, then up to twenty spread through the file
395
395
  pub row_count: usize,
396
396
  }
397
397
  ```
package/llms.txt CHANGED
@@ -31,7 +31,8 @@ resolve.
31
31
  - `parse_csv(bytes, mapping)` / `parse_json(bytes, mapping)` / `parse_parquet(bytes, mapping)` - full parse with an explicit `CsvMapping`. There is no header auto-detection.
32
32
  - `stdf_test_names(bytes)` / `atdf_test_names(bytes)` - fast scan: test definitions + die count, no per-die accumulation.
33
33
  - `parse_stdf_filtered(bytes, selected)` / `parse_atdf_filtered(bytes, selected)` - full parse that accumulates values only for the selected test numbers.
34
- - `parquet_headers(bytes)` - schema, sample rows and per-column types, for a mapping UI. There is deliberately no `csv_headers`/`json_headers` in the WASM API: read a text header row in plain JS.
34
+ - `parquet_headers(bytes)` - schema, sample rows (the first five, then the first row of up to twenty later row groups) and per-column types, for a mapping UI.
35
+ - `parquet_distinct_count(bytes, columns, blankIsAValue)` - distinct combinations of the named columns, read as a projection. `blankIsAValue: false` skips rows blank in every named column (a wafer count); `true` counts them as a value (is this column constant?). There is deliberately no `csv_headers`/`json_headers` in the WASM API: read a text header row in plain JS.
35
36
  - `stdf_file_meta(bytes)` / `atdf_file_meta(bytes)` - lot metadata, wafer count, first/last timestamps and site count, without a full parse.
36
37
 
37
38
  ## Traps
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@wafertools/testdata-parser",
3
3
  "type": "module",
4
4
  "description": "Rust/WASM parsers for semiconductor test data formats (STDF, ATDF, CSV, JSON, Parquet)",
5
- "version": "0.14.0",
5
+ "version": "0.15.0",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
@@ -240,7 +240,7 @@ export function init(): void;
240
240
  /**
241
241
  * `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
242
242
  */
243
- export function parquet_distinct_count(bytes: Uint8Array, columns: string[]): number;
243
+ export function parquet_distinct_count(bytes: Uint8Array, columns: string[], blank_is_a_value: boolean): number;
244
244
 
245
245
  export function parquet_headers(bytes: Uint8Array): ParquetHeadersResult;
246
246
 
@@ -269,7 +269,7 @@ export interface InitOutput {
269
269
  readonly atdf_file_meta: (a: number, b: number) => [number, number, number];
270
270
  readonly atdf_test_names: (a: number, b: number) => [number, number, number];
271
271
  readonly init: () => void;
272
- readonly parquet_distinct_count: (a: number, b: number, c: any) => [number, number, number];
272
+ readonly parquet_distinct_count: (a: number, b: number, c: any, d: number) => [number, number, number];
273
273
  readonly parquet_headers: (a: number, b: number) => [number, number, number];
274
274
  readonly parse_atdf: (a: number, b: number) => [number, number, number, number];
275
275
  readonly parse_atdf_filtered: (a: number, b: number, c: any) => [number, number, number, number];
@@ -36,12 +36,13 @@ export function init() {
36
36
  * `columns` is a JS string array. See `parquet_distinct_count_from_bytes`.
37
37
  * @param {Uint8Array} bytes
38
38
  * @param {string[]} columns
39
+ * @param {boolean} blank_is_a_value
39
40
  * @returns {number}
40
41
  */
41
- export function parquet_distinct_count(bytes, columns) {
42
+ export function parquet_distinct_count(bytes, columns, blank_is_a_value) {
42
43
  const ptr0 = passArray8ToWasm0(bytes, wasm.__wbindgen_malloc);
43
44
  const len0 = WASM_VECTOR_LEN;
44
- const ret = wasm.parquet_distinct_count(ptr0, len0, columns);
45
+ const ret = wasm.parquet_distinct_count(ptr0, len0, columns, blank_is_a_value);
45
46
  if (ret[2]) {
46
47
  throw takeFromExternrefTable0(ret[1]);
47
48
  }
Binary file