@schmock/faker 2.2.3 → 2.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +45 -0
  3. package/dist/constants.d.ts +82 -1
  4. package/dist/constants.d.ts.map +1 -1
  5. package/dist/field-mappings.d.ts.map +1 -1
  6. package/dist/field-name-matcher.d.ts +19 -0
  7. package/dist/field-name-matcher.d.ts.map +1 -1
  8. package/dist/index.d.ts +3 -0
  9. package/dist/index.d.ts.map +1 -1
  10. package/dist/index.js +4 -1
  11. package/dist/jsf-config.d.ts +27 -2
  12. package/dist/jsf-config.d.ts.map +1 -1
  13. package/dist/output-limits.d.ts +3 -0
  14. package/dist/output-limits.d.ts.map +1 -0
  15. package/dist/overrides.d.ts +2 -1
  16. package/dist/overrides.d.ts.map +1 -1
  17. package/dist/schema-children.d.ts +14 -0
  18. package/dist/schema-children.d.ts.map +1 -0
  19. package/dist/schema-enhancement.d.ts +1 -0
  20. package/dist/schema-enhancement.d.ts.map +1 -1
  21. package/dist/validation.d.ts +24 -5
  22. package/dist/validation.d.ts.map +1 -1
  23. package/package.json +19 -9
  24. package/dist/constants.js +0 -8
  25. package/dist/field-mappings.js +0 -1107
  26. package/dist/field-name-matcher.js +0 -149
  27. package/dist/jsf-config.js +0 -35
  28. package/dist/overrides.js +0 -131
  29. package/dist/schema-enhancement.js +0 -89
  30. package/dist/test-utils.d.ts +0 -57
  31. package/dist/test-utils.d.ts.map +0 -1
  32. package/dist/test-utils.js +0 -271
  33. package/dist/validation.js +0 -300
  34. package/src/advanced-features.test.ts +0 -912
  35. package/src/audit-faker-args.test.ts +0 -97
  36. package/src/constants.d.ts.map +0 -1
  37. package/src/constants.ts +0 -9
  38. package/src/data-quality.test.ts +0 -415
  39. package/src/error-handling.test.ts +0 -504
  40. package/src/field-mappings.test.ts +0 -351
  41. package/src/field-mappings.ts +0 -1141
  42. package/src/field-name-matcher.test.ts +0 -292
  43. package/src/field-name-matcher.ts +0 -188
  44. package/src/index.d.ts.map +0 -1
  45. package/src/index.test.ts +0 -1218
  46. package/src/index.ts +0 -168
  47. package/src/integration.test.ts +0 -634
  48. package/src/jsf-config.d.ts.map +0 -1
  49. package/src/jsf-config.ts +0 -48
  50. package/src/overrides.d.ts.map +0 -1
  51. package/src/overrides.property.test.ts +0 -172
  52. package/src/overrides.test.ts +0 -38
  53. package/src/overrides.ts +0 -176
  54. package/src/performance.test.ts +0 -494
  55. package/src/plugin-integration.test.ts +0 -575
  56. package/src/post-process.test.ts +0 -151
  57. package/src/real-world.test.ts +0 -636
  58. package/src/schema-enhancement.d.ts.map +0 -1
  59. package/src/schema-enhancement.test.ts +0 -216
  60. package/src/schema-enhancement.ts +0 -128
  61. package/src/steps/deterministic-seeds.steps.ts +0 -35
  62. package/src/steps/faker-plugin.steps.ts +0 -109
  63. package/src/test-utils.ts +0 -365
  64. package/src/validation.d.ts.map +0 -1
  65. package/src/validation.ts +0 -417
@@ -1,292 +0,0 @@
1
- import { describe, expect, it } from "vitest";
2
- import {
3
- findBestMapping,
4
- scoreMatch,
5
- tokenizeFieldName,
6
- } from "./field-name-matcher";
7
-
8
- describe("tokenizeFieldName", () => {
9
- it("splits camelCase", () => {
10
- expect(tokenizeFieldName("userFirstName")).toEqual([
11
- "user",
12
- "first",
13
- "name",
14
- ]);
15
- });
16
-
17
- it("splits snake_case", () => {
18
- expect(tokenizeFieldName("created_at")).toEqual(["created", "at"]);
19
- });
20
-
21
- it("splits kebab-case", () => {
22
- expect(tokenizeFieldName("user-name")).toEqual(["user", "name"]);
23
- });
24
-
25
- it("handles consecutive uppercase (HTMLParser)", () => {
26
- expect(tokenizeFieldName("HTMLParser")).toEqual(["html", "parser"]);
27
- });
28
-
29
- it("handles is_active", () => {
30
- expect(tokenizeFieldName("is_active")).toEqual(["is", "active"]);
31
- });
32
-
33
- it("handles single word", () => {
34
- expect(tokenizeFieldName("email")).toEqual(["email"]);
35
- });
36
-
37
- it("handles uppercase single word", () => {
38
- expect(tokenizeFieldName("UUID")).toEqual(["uuid"]);
39
- });
40
-
41
- it("handles mixed formats", () => {
42
- expect(tokenizeFieldName("userEmail_address")).toEqual([
43
- "user",
44
- "email",
45
- "address",
46
- ]);
47
- });
48
- });
49
-
50
- describe("scoreMatch", () => {
51
- it("returns 1.0 for exact match", () => {
52
- expect(scoreMatch(["email"], ["email"])).toBe(1.0);
53
- });
54
-
55
- it("returns 1.0 for exact multi-token match", () => {
56
- expect(scoreMatch(["first", "name"], ["first_name"])).toBe(1.0);
57
- });
58
-
59
- it("returns 0.7 for substring match with low keyword coverage", () => {
60
- // "email" is 1/3 tokens → low coverage (0.65) but substring match (0.7) wins
61
- expect(scoreMatch(["user", "email", "address"], ["email"])).toBe(0.7);
62
- });
63
-
64
- it("returns 0.9 when all keyword tokens found with high coverage", () => {
65
- // ["created", "at"] are both found in ["user", "created", "at"] → 2/3 coverage > 50%
66
- expect(scoreMatch(["user", "created", "at"], ["created_at"])).toBe(0.9);
67
- });
68
-
69
- it("returns 0.8 when field ends with keyword", () => {
70
- // ["name"] is at the end of ["display", "name"] — ends-with score = 0.8
71
- // coverage is 1/2 = 0.5, not > 0.5, so "all found" gives 0.65
72
- // ends-with wins at 0.8
73
- expect(scoreMatch(["display", "name"], ["name"])).toBe(0.8);
74
- });
75
-
76
- it("returns 0.7 for substring match", () => {
77
- expect(scoreMatch(["myemailfield"], ["email"])).toBe(0.7);
78
- });
79
-
80
- it("returns 0 for no match", () => {
81
- expect(scoreMatch(["foo", "bar"], ["email"])).toBe(0);
82
- });
83
- });
84
-
85
- describe("findBestMapping", () => {
86
- it("maps email field", () => {
87
- const result = findBestMapping("email", { type: "string" });
88
- expect(result).toBeDefined();
89
- expect(result?.mapping.fakerMethod).toBe("internet.email");
90
- });
91
-
92
- it("maps userEmail field", () => {
93
- const result = findBestMapping("userEmail", { type: "string" });
94
- expect(result).toBeDefined();
95
- expect(result?.mapping.fakerMethod).toBe("internet.email");
96
- });
97
-
98
- it("maps firstName field", () => {
99
- const result = findBestMapping("firstName", { type: "string" });
100
- expect(result).toBeDefined();
101
- expect(result?.mapping.fakerMethod).toBe("person.firstName");
102
- });
103
-
104
- it("maps first_name field", () => {
105
- const result = findBestMapping("first_name", { type: "string" });
106
- expect(result).toBeDefined();
107
- expect(result?.mapping.fakerMethod).toBe("person.firstName");
108
- });
109
-
110
- it("maps createdAt to date.recent", () => {
111
- const result = findBestMapping("createdAt", { type: "string" });
112
- expect(result).toBeDefined();
113
- expect(result?.mapping.fakerMethod).toBe("date.recent");
114
- expect(result?.mapping.format).toBe("date-time");
115
- });
116
-
117
- it("maps city field", () => {
118
- const result = findBestMapping("city", { type: "string" });
119
- expect(result).toBeDefined();
120
- expect(result?.mapping.fakerMethod).toBe("location.city");
121
- });
122
-
123
- it("maps url field", () => {
124
- const result = findBestMapping("url", { type: "string" });
125
- expect(result).toBeDefined();
126
- expect(result?.mapping.fakerMethod).toBe("internet.url");
127
- });
128
-
129
- it("maps avatar field", () => {
130
- const result = findBestMapping("avatar", { type: "string" });
131
- expect(result).toBeDefined();
132
- expect(result?.mapping.fakerMethod).toBe("image.avatar");
133
- });
134
-
135
- it("maps latitude field to number", () => {
136
- const result = findBestMapping("latitude", { type: "number" });
137
- expect(result).toBeDefined();
138
- expect(result?.mapping.fakerMethod).toBe("location.latitude");
139
- });
140
-
141
- it("maps description field", () => {
142
- const result = findBestMapping("description", { type: "string" });
143
- expect(result).toBeDefined();
144
- expect(result?.mapping.fakerMethod).toBe("lorem.paragraph");
145
- });
146
-
147
- it("maps title field", () => {
148
- const result = findBestMapping("title", { type: "string" });
149
- expect(result).toBeDefined();
150
- expect(result?.mapping.fakerMethod).toBe("lorem.sentence");
151
- });
152
-
153
- it("maps country field", () => {
154
- const result = findBestMapping("country", { type: "string" });
155
- expect(result).toBeDefined();
156
- expect(result?.mapping.fakerMethod).toBe("location.country");
157
- });
158
-
159
- it("maps isActive boolean with probability", () => {
160
- const result = findBestMapping("isActive", { type: "boolean" });
161
- expect(result).toBeDefined();
162
- expect(result?.mapping.trueProbability).toBe(0.9);
163
- });
164
-
165
- it("maps is_deleted boolean with low probability", () => {
166
- const result = findBestMapping("is_deleted", { type: "boolean" });
167
- expect(result).toBeDefined();
168
- expect(result?.mapping.trueProbability).toBe(0.05);
169
- });
170
-
171
- describe("ID suffix detection", () => {
172
- it("maps userId to UUID", () => {
173
- const result = findBestMapping("userId", { type: "string" });
174
- expect(result).toBeDefined();
175
- expect(result?.mapping.fakerMethod).toBe("string.uuid");
176
- });
177
-
178
- it("maps order_id to UUID", () => {
179
- const result = findBestMapping("order_id", { type: "string" });
180
- expect(result).toBeDefined();
181
- expect(result?.mapping.fakerMethod).toBe("string.uuid");
182
- });
183
-
184
- it("maps parent_id to UUID", () => {
185
- const result = findBestMapping("parent_id", { type: "string" });
186
- expect(result).toBeDefined();
187
- expect(result?.mapping.fakerMethod).toBe("string.uuid");
188
- });
189
-
190
- it("does not map single id without format:uuid", () => {
191
- const result = findBestMapping("id", { type: "string" });
192
- // 'id' alone shouldn't trigger UUID — it's a single token so suffix rule doesn't apply
193
- // But it could still match something else. Let's just check it doesn't falsely match
194
- if (result) {
195
- // Could match 'id' in some mapping, that's OK
196
- expect(result.score).toBeGreaterThan(0);
197
- }
198
- });
199
-
200
- it("does not map Id suffix on number fields", () => {
201
- const result = findBestMapping("userId", { type: "number" });
202
- // Number type should not get UUID mapping
203
- if (result) {
204
- expect(result.mapping.fakerMethod).not.toBe("string.uuid");
205
- }
206
- });
207
- });
208
-
209
- describe("skip conditions", () => {
210
- it("skips when schema has pattern", () => {
211
- const result = findBestMapping("email", {
212
- type: "string",
213
- pattern: "^[a-z]+$",
214
- });
215
- expect(result).toBeUndefined();
216
- });
217
-
218
- it("skips when schema has enum", () => {
219
- const result = findBestMapping("email", {
220
- type: "string",
221
- enum: ["a@b.com", "c@d.com"],
222
- });
223
- expect(result).toBeUndefined();
224
- });
225
-
226
- it("skips when schema already has faker", () => {
227
- const schema = { type: "string" as const, faker: "lorem.word" } as any;
228
- const result = findBestMapping("email", schema);
229
- expect(result).toBeUndefined();
230
- });
231
-
232
- // Regression: name-based mappings (e.g. lorem.word for 'label') don't
233
- // honor JSON Schema length constraints, so they'd produce out-of-range
234
- // strings ~20% of the time when the schema asked for a specific length.
235
- // Skip the mapping and let json-schema-faker generate a length-respecting
236
- // string instead. Mirrors the numeric constraint skip in the loop below.
237
- it("skips string mapping when schema has minLength", () => {
238
- const result = findBestMapping("label", {
239
- type: "string",
240
- minLength: 3,
241
- });
242
- expect(result).toBeUndefined();
243
- });
244
-
245
- it("skips string mapping when schema has maxLength", () => {
246
- const result = findBestMapping("label", {
247
- type: "string",
248
- maxLength: 20,
249
- });
250
- expect(result).toBeUndefined();
251
- });
252
-
253
- it("skips string mapping when both length constraints are set", () => {
254
- const result = findBestMapping("label", {
255
- type: "string",
256
- minLength: 3,
257
- maxLength: 20,
258
- });
259
- expect(result).toBeUndefined();
260
- });
261
-
262
- it("does not skip non-string types when minLength happens to be set", () => {
263
- // minLength is a string-only keyword; on a non-string schema it's
264
- // meaningless. The number mapping for 'age' should still apply.
265
- const result = findBestMapping("age", {
266
- type: "number",
267
- minLength: 3, // nonsensical on a number, but shouldn't block the mapping
268
- } as any);
269
- expect(result).toBeDefined();
270
- });
271
- });
272
-
273
- it("does not map unrecognized fields", () => {
274
- const result = findBestMapping("randomFieldXYZ123", { type: "string" });
275
- expect(result).toBeUndefined();
276
- });
277
-
278
- it("respects type constraints", () => {
279
- // latitude mapping requires number type
280
- const result = findBestMapping("latitude", { type: "string" });
281
- expect(result).toBeUndefined();
282
- });
283
-
284
- it("maps format:uuid even without name match", () => {
285
- const result = findBestMapping("someField", {
286
- type: "string",
287
- format: "uuid",
288
- });
289
- expect(result).toBeDefined();
290
- expect(result?.mapping.fakerMethod).toBe("string.uuid");
291
- });
292
- });
@@ -1,188 +0,0 @@
1
- import type { JSONSchema7 } from "json-schema";
2
- import type { FieldMapping } from "./field-mappings.js";
3
- import { ALL_FIELD_MAPPINGS } from "./field-mappings.js";
4
-
5
- /**
6
- * Split a field name (camelCase, snake_case, kebab-case) into lowercase tokens.
7
- * "userFirstName" → ["user", "first", "name"]
8
- * "created_at" → ["created", "at"]
9
- * "HTMLParser" → ["html", "parser"]
10
- * "is_active" → ["is", "active"]
11
- */
12
- export function tokenizeFieldName(name: string): string[] {
13
- // Split on _ and -
14
- const parts = name.split(/[_-]/);
15
- const tokens: string[] = [];
16
-
17
- for (const part of parts) {
18
- if (!part) continue;
19
- // Split camelCase and consecutive uppercase (e.g., HTMLParser → HTML, Parser)
20
- const camelTokens = part
21
- .replace(/([A-Z]+)([A-Z][a-z])/g, "$1_$2")
22
- .replace(/([a-z0-9])([A-Z])/g, "$1_$2")
23
- .split("_");
24
-
25
- for (const t of camelTokens) {
26
- if (t) tokens.push(t.toLowerCase());
27
- }
28
- }
29
-
30
- return tokens;
31
- }
32
-
33
- /**
34
- * Score how well a set of keyword tokens matches field name tokens.
35
- * Returns 0-1:
36
- * 1.0 — exact full match (joined tokens equal joined keywords)
37
- * 0.9 — all keyword tokens found in field tokens with high coverage (>50%)
38
- * 0.8 — field ends with keyword tokens
39
- * 0.7 — substring match (keyword appears in joined field name)
40
- * 0.65 — all keyword tokens found but low coverage (<=50%)
41
- */
42
- export function scoreMatch(fieldTokens: string[], keywords: string[]): number {
43
- const fieldJoined = fieldTokens.join("");
44
-
45
- let bestScore = 0;
46
-
47
- // Try each keyword variant, keep the highest score
48
- for (const keyword of keywords) {
49
- const kwTokens = tokenizeFieldName(keyword);
50
- const kwJoined = kwTokens.join("");
51
-
52
- // Exact match: joined tokens are identical
53
- if (fieldJoined === kwJoined) return 1.0;
54
-
55
- // All keyword tokens present in field tokens with high coverage
56
- if (
57
- kwTokens.length > 0 &&
58
- kwTokens.every((kt) => fieldTokens.includes(kt))
59
- ) {
60
- const coverage = kwTokens.length / fieldTokens.length;
61
- const score = coverage > 0.5 ? 0.9 : 0.65;
62
- bestScore = Math.max(bestScore, score);
63
- }
64
-
65
- // Field ends with keyword tokens
66
- if (kwTokens.length > 0 && kwTokens.length <= fieldTokens.length) {
67
- const tail = fieldTokens.slice(-kwTokens.length);
68
- if (tail.every((t, i) => t === kwTokens[i])) {
69
- bestScore = Math.max(bestScore, 0.8);
70
- }
71
- }
72
-
73
- // Substring match: keyword joined appears in field joined
74
- if (kwJoined.length >= 3 && fieldJoined.includes(kwJoined)) {
75
- bestScore = Math.max(bestScore, 0.7);
76
- }
77
- }
78
-
79
- return bestScore;
80
- }
81
-
82
- interface MatchResult {
83
- mapping: FieldMapping;
84
- score: number;
85
- }
86
-
87
- /**
88
- * Find the best field mapping for a given field name and schema.
89
- * Returns the highest-scoring match above its threshold, or undefined.
90
- */
91
- export function findBestMapping(
92
- fieldName: string,
93
- schema: JSONSchema7,
94
- mappings: FieldMapping[] = ALL_FIELD_MAPPINGS,
95
- ): MatchResult | undefined {
96
- const schemaAny = schema as Record<string, unknown>;
97
- const schemaType = typeof schema.type === "string" ? schema.type : undefined;
98
-
99
- // Priority: format:uuid always maps to string.uuid
100
- if (schemaType === "string" && schema.format === "uuid") {
101
- return {
102
- mapping: {
103
- keywords: ["uuid"],
104
- fakerMethod: "string.uuid",
105
- schemaType: "string",
106
- minScore: 0.5,
107
- },
108
- score: 1.0,
109
- };
110
- }
111
-
112
- // Skip if schema already has pattern, enum, faker, or $ref constraint
113
- // Also skip string mappings when minLength/maxLength is set — name-based
114
- // mappings (e.g. lorem.word for "label") don't honor JSON Schema length
115
- // constraints, so they'd produce out-of-range strings ~20% of the time.
116
- // Mirrors the numeric constraint skip just below.
117
- if (
118
- schemaAny.pattern ||
119
- schemaAny.enum ||
120
- schemaAny.faker ||
121
- schemaAny.$ref ||
122
- (schemaType === "string" &&
123
- (schema.minLength !== undefined || schema.maxLength !== undefined))
124
- ) {
125
- return undefined;
126
- }
127
-
128
- // Skip numeric faker mapping when schema already constrains the range
129
- const hasNumericConstraints =
130
- schema.minimum !== undefined ||
131
- schema.maximum !== undefined ||
132
- schema.exclusiveMinimum !== undefined ||
133
- schema.exclusiveMaximum !== undefined;
134
-
135
- const tokens = tokenizeFieldName(fieldName);
136
-
137
- let best: MatchResult | undefined;
138
-
139
- for (const mapping of mappings) {
140
- // Skip numeric faker mappings when schema has explicit constraints
141
- if (
142
- hasNumericConstraints &&
143
- (mapping.schemaType === "number" || mapping.schemaType === "integer")
144
- ) {
145
- continue;
146
- }
147
-
148
- // Skip faker mappings for object/array schemas — they generate primitive values
149
- if (schemaType === "object" || schemaType === "array") {
150
- continue;
151
- }
152
-
153
- // Check schema type constraint
154
- if (mapping.schemaType && schemaType && mapping.schemaType !== schemaType) {
155
- const isNumeric =
156
- (mapping.schemaType === "number" && schemaType === "integer") ||
157
- (mapping.schemaType === "integer" && schemaType === "number");
158
- if (!isNumeric) continue;
159
- }
160
-
161
- const score = scoreMatch(tokens, mapping.keywords);
162
- if (score >= mapping.minScore && (!best || score > best.score)) {
163
- best = { mapping, score };
164
- }
165
- }
166
-
167
- // ID suffix detection: fields ending in Id/_id with string type → UUID
168
- // Only if no better match was found or the existing match is weak
169
- if (schemaType === "string") {
170
- const lastToken = tokens[tokens.length - 1];
171
- if (lastToken === "id" && tokens.length > 1) {
172
- const idScore = 0.8;
173
- if (!best || best.score < idScore) {
174
- best = {
175
- mapping: {
176
- keywords: ["id"],
177
- fakerMethod: "string.uuid",
178
- schemaType: "string",
179
- minScore: 0.5,
180
- },
181
- score: idScore,
182
- };
183
- }
184
- }
185
- }
186
-
187
- return best;
188
- }
@@ -1 +0,0 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["index.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAO/C,MAAM,WAAW,uBAAuB;IACtC,MAAM,EAAE,WAAW,CAAC;IACpB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,SAAS,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IAChC,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAChC,KAAK,CAAC,EAAE,GAAG,CAAC;IACZ,KAAK,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC/B,IAAI,CAAC,EAAE,MAAM,CAAC;CACf;AAED,MAAM,WAAW,kBAAkB;IACjC,MAAM,EAAE,WAAW,CAAC;IACpB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,SAAS,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IAChC,IAAI,CAAC,EAAE,MAAM,CAAC;CACf;AAED,wBAAgB,WAAW,CAAC,OAAO,EAAE,kBAAkB,GAAG,OAAO,CAAC,MAAM,CA+CvE;AAED,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,uBAAuB,GAAG,GAAG,CA6CxE"}