@hyperfixi/patterns-reference 3.1.1 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +186 -1
- package/README.md +47 -36
- package/data/engine-verification.json +58 -65
- package/data/patterns.db +0 -0
- package/data/patterns.db.stamp +1 -0
- package/dist/api/index.d.mts +2 -2
- package/dist/api/index.d.ts +2 -2
- package/dist/api/index.js +198 -98
- package/dist/api/index.mjs +197 -100
- package/dist/{index-6GkHj5yJ.d.mts → index-CoDUfq2P.d.mts} +39 -3
- package/dist/{index-6GkHj5yJ.d.ts → index-CoDUfq2P.d.ts} +39 -3
- package/dist/index.d.mts +87 -13
- package/dist/index.d.ts +87 -13
- package/dist/index.js +261 -144
- package/dist/index.mjs +258 -146
- package/dist/{llm-B5nGz8V1.d.mts → llm-CCCSw-yp.d.ts} +21 -10
- package/dist/{llm-rEdJQScF.d.ts → llm-PxnriEZ_.d.mts} +21 -10
- package/dist/sync/index.d.mts +1 -1
- package/dist/sync/index.d.ts +1 -1
- package/dist/sync/index.js +4 -1
- package/dist/sync/index.mjs +4 -1
- package/package.json +16 -16
- package/src/adapters/llm-adapter.ts +64 -42
- package/src/api/engine-filter.ts +37 -0
- package/src/api/llm.ts +54 -64
- package/src/api/patterns.ts +92 -30
- package/src/api/roles.ts +6 -4
- package/src/api/translations.ts +34 -27
- package/src/database/connection.ts +16 -4
- package/src/html-snippets.ts +39 -10
- package/src/index.ts +12 -2
- package/src/registry/patterns-provider.ts +3 -1
- package/src/sync/db-stamp.ts +7 -4
- package/src/sync/markup-attributes.ts +15 -0
- package/src/sync/verify-parses.ts +42 -0
- package/src/types/better-sqlite3.d.ts +2 -0
- package/src/types/index.ts +44 -2
- package/src/sync/span-mask.ts +0 -166
- package/src/sync/translation-checks.ts +0 -282
|
@@ -1,282 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Post-translation correctness checks. Each check takes the source raw_code
|
|
3
|
-
* and a translated string and returns a pass/fail with a diagnostic message.
|
|
4
|
-
*
|
|
5
|
-
* Designed to catch the artifact classes documented in the patterns-reference
|
|
6
|
-
* fix plan:
|
|
7
|
-
* A: HTML element reordering (caught by `htmlParityCheck`)
|
|
8
|
-
* B/C/D: literal/URL/directive bleeding (caught by `literalPreservationCheck`)
|
|
9
|
-
* E: silent truncation, e.g. zh dropping `to <form/>` (caught by `tokenCountCheck`)
|
|
10
|
-
*/
|
|
11
|
-
|
|
12
|
-
import { maskSpans } from './span-mask';
|
|
13
|
-
|
|
14
|
-
export interface CheckResult {
|
|
15
|
-
valid: boolean;
|
|
16
|
-
error?: string;
|
|
17
|
-
}
|
|
18
|
-
|
|
19
|
-
// =============================================================================
|
|
20
|
-
// HTML well-formedness parity
|
|
21
|
-
// =============================================================================
|
|
22
|
-
|
|
23
|
-
/**
|
|
24
|
-
* Compare the multiset of HTML tags between source and translation. Tag names
|
|
25
|
-
* (case-insensitive) and counts must match. Attribute values are NOT compared
|
|
26
|
-
* because masking + unmasking preserves them verbatim by construction; the
|
|
27
|
-
* concern here is structural.
|
|
28
|
-
*/
|
|
29
|
-
export function htmlParityCheck(source: string, translated: string): CheckResult {
|
|
30
|
-
const sourceTags = extractTags(source);
|
|
31
|
-
const translatedTags = extractTags(translated);
|
|
32
|
-
|
|
33
|
-
if (sourceTags.length === 0 && translatedTags.length === 0) {
|
|
34
|
-
return { valid: true };
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
const sourceCount = countTags(sourceTags);
|
|
38
|
-
const translatedCount = countTags(translatedTags);
|
|
39
|
-
|
|
40
|
-
for (const [name, n] of sourceCount) {
|
|
41
|
-
const m = translatedCount.get(name) || 0;
|
|
42
|
-
if (m !== n) {
|
|
43
|
-
return {
|
|
44
|
-
valid: false,
|
|
45
|
-
error: `tag <${name}> count mismatch: source has ${n}, translation has ${m}`,
|
|
46
|
-
};
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
for (const [name, m] of translatedCount) {
|
|
50
|
-
if (!sourceCount.has(name)) {
|
|
51
|
-
return {
|
|
52
|
-
valid: false,
|
|
53
|
-
error: `tag <${name}> appears in translation but not in source`,
|
|
54
|
-
};
|
|
55
|
-
}
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
// Check nesting order: at no point during a left-to-right scan should the
|
|
59
|
-
// running open/close balance per tag name go negative. That would mean a
|
|
60
|
-
// `</tag>` appears before its matching `<tag>` — exactly the artifact A
|
|
61
|
-
// failure mode where SOV reordering produced `World</span> を <span>Hello`.
|
|
62
|
-
const orderError = checkNestingOrder(translatedTags);
|
|
63
|
-
if (orderError) {
|
|
64
|
-
return { valid: false, error: orderError };
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
return { valid: true };
|
|
68
|
-
}
|
|
69
|
-
|
|
70
|
-
interface TagToken {
|
|
71
|
-
name: string;
|
|
72
|
-
kind: 'open' | 'close' | 'self';
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
function extractTags(input: string): TagToken[] {
|
|
76
|
-
const tags: TagToken[] = [];
|
|
77
|
-
const re = /<\/?([a-zA-Z][\w-]*)(?:\s[^>]*)?>/g;
|
|
78
|
-
let m: RegExpExecArray | null;
|
|
79
|
-
while ((m = re.exec(input)) !== null) {
|
|
80
|
-
const isClose = m[0].startsWith('</');
|
|
81
|
-
// Self-closing detection: the matched tag string ends with `/>`. Using a
|
|
82
|
-
// separate `(\/)?` capture group fails when the inner [^>]* greedily
|
|
83
|
-
// consumes the trailing `/` (it does, because `/` ≠ `>`).
|
|
84
|
-
const isSelf = !isClose && m[0].endsWith('/>');
|
|
85
|
-
tags.push({
|
|
86
|
-
name: m[1].toLowerCase(),
|
|
87
|
-
kind: isClose ? 'close' : isSelf ? 'self' : 'open',
|
|
88
|
-
});
|
|
89
|
-
}
|
|
90
|
-
return tags;
|
|
91
|
-
}
|
|
92
|
-
|
|
93
|
-
function countTags(tags: TagToken[]): Map<string, number> {
|
|
94
|
-
const counts = new Map<string, number>();
|
|
95
|
-
for (const t of tags) {
|
|
96
|
-
counts.set(t.name, (counts.get(t.name) || 0) + 1);
|
|
97
|
-
}
|
|
98
|
-
return counts;
|
|
99
|
-
}
|
|
100
|
-
|
|
101
|
-
/**
|
|
102
|
-
* Walk the tag list left-to-right and verify each closing tag has a prior
|
|
103
|
-
* matching opening tag (per name). Returns an error message if a close
|
|
104
|
-
* appears before an open, otherwise null.
|
|
105
|
-
*/
|
|
106
|
-
function checkNestingOrder(tags: TagToken[]): string | null {
|
|
107
|
-
const open = new Map<string, number>();
|
|
108
|
-
for (const t of tags) {
|
|
109
|
-
if (t.kind === 'self') continue;
|
|
110
|
-
if (t.kind === 'open') {
|
|
111
|
-
open.set(t.name, (open.get(t.name) || 0) + 1);
|
|
112
|
-
} else {
|
|
113
|
-
const count = open.get(t.name) || 0;
|
|
114
|
-
if (count <= 0) {
|
|
115
|
-
return `</${t.name}> appears before its matching <${t.name}>`;
|
|
116
|
-
}
|
|
117
|
-
open.set(t.name, count - 1);
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
for (const [name, count] of open) {
|
|
121
|
-
if (count !== 0) {
|
|
122
|
-
return `<${name}> not closed (${count} unclosed)`;
|
|
123
|
-
}
|
|
124
|
-
}
|
|
125
|
-
return null;
|
|
126
|
-
}
|
|
127
|
-
|
|
128
|
-
// =============================================================================
|
|
129
|
-
// Literal preservation
|
|
130
|
-
// =============================================================================
|
|
131
|
-
|
|
132
|
-
/**
|
|
133
|
-
* Every string literal, URL, template literal, and directive from the source
|
|
134
|
-
* must appear verbatim in the translation. Catches "translator bled into a
|
|
135
|
-
* string" artifacts (he turning `'Got it!'` into `'Got זה!'`, ms turning
|
|
136
|
-
* `body:` into `badan:`, etc.).
|
|
137
|
-
*
|
|
138
|
-
* Bracket expressions are intentionally excluded: their inner string literals
|
|
139
|
-
* are already covered by the string-* spans, and bracket interiors may contain
|
|
140
|
-
* identifiers that are legitimately translatable (e.g. `[key is 'Escape']`
|
|
141
|
-
* where `key` and `is` may be translated, while `'Escape'` is not).
|
|
142
|
-
*/
|
|
143
|
-
export function literalPreservationCheck(source: string, translated: string): CheckResult {
|
|
144
|
-
const { spans } = maskSpans(source);
|
|
145
|
-
const protectedKinds = new Set([
|
|
146
|
-
'string-single',
|
|
147
|
-
'string-double',
|
|
148
|
-
'string-template',
|
|
149
|
-
'url',
|
|
150
|
-
'directive',
|
|
151
|
-
'js-block',
|
|
152
|
-
]);
|
|
153
|
-
|
|
154
|
-
for (const span of spans) {
|
|
155
|
-
if (!protectedKinds.has(span.kind)) continue;
|
|
156
|
-
if (!translated.includes(span.original)) {
|
|
157
|
-
return {
|
|
158
|
-
valid: false,
|
|
159
|
-
error: `${span.kind} literal not preserved: ${truncate(span.original, 50)}`,
|
|
160
|
-
};
|
|
161
|
-
}
|
|
162
|
-
}
|
|
163
|
-
return { valid: true };
|
|
164
|
-
}
|
|
165
|
-
|
|
166
|
-
function truncate(s: string, n: number): string {
|
|
167
|
-
return s.length > n ? s.slice(0, n - 1) + '…' : s;
|
|
168
|
-
}
|
|
169
|
-
|
|
170
|
-
// =============================================================================
|
|
171
|
-
// Token count guard (truncation detector)
|
|
172
|
-
// =============================================================================
|
|
173
|
-
|
|
174
|
-
/**
|
|
175
|
-
* Whitespace-token-count ratio. If the translation is dramatically shorter
|
|
176
|
-
* than the source, it likely silently dropped trailing roles (artifact E).
|
|
177
|
-
*
|
|
178
|
-
* Tolerances are language-family specific. CJK and SOV non-CJK can compress
|
|
179
|
-
* by ~30% legitimately (particles, agglutinative suffixes), so the floors
|
|
180
|
-
* are looser than for Romance/Germanic. The truncation cases observed in
|
|
181
|
-
* Chinese drop ratios to 0.4-0.5; these bounds catch that without flagging
|
|
182
|
-
* normal compression.
|
|
183
|
-
*/
|
|
184
|
-
export function tokenCountCheck(source: string, translated: string, language: string): CheckResult {
|
|
185
|
-
const sourceTokens = countWhitespaceTokens(source);
|
|
186
|
-
const translatedTokens = countWhitespaceTokens(translated);
|
|
187
|
-
|
|
188
|
-
if (sourceTokens === 0) {
|
|
189
|
-
return { valid: translatedTokens === 0 };
|
|
190
|
-
}
|
|
191
|
-
|
|
192
|
-
const ratio = translatedTokens / sourceTokens;
|
|
193
|
-
const [floor, ceiling] = getTolerance(language);
|
|
194
|
-
|
|
195
|
-
if (ratio < floor) {
|
|
196
|
-
return {
|
|
197
|
-
valid: false,
|
|
198
|
-
error: `truncation suspected: ${translatedTokens}/${sourceTokens} tokens (ratio ${ratio.toFixed(2)} < floor ${floor})`,
|
|
199
|
-
};
|
|
200
|
-
}
|
|
201
|
-
if (ratio > ceiling) {
|
|
202
|
-
return {
|
|
203
|
-
valid: false,
|
|
204
|
-
error: `expansion suspected: ${translatedTokens}/${sourceTokens} tokens (ratio ${ratio.toFixed(2)} > ceiling ${ceiling})`,
|
|
205
|
-
};
|
|
206
|
-
}
|
|
207
|
-
return { valid: true };
|
|
208
|
-
}
|
|
209
|
-
|
|
210
|
-
function countWhitespaceTokens(s: string): number {
|
|
211
|
-
return s.split(/\s+/).filter(Boolean).length;
|
|
212
|
-
}
|
|
213
|
-
|
|
214
|
-
function getTolerance(language: string): [number, number] {
|
|
215
|
-
// [floor, ceiling]. The primary purpose of the truncation guard is to catch
|
|
216
|
-
// *drops* (translation silently losing trailing roles), so the floor is the
|
|
217
|
-
// load-bearing bound. Ceilings are intentionally generous because legitimate
|
|
218
|
-
// SOV/agglutinative translations regularly add particles per substantive
|
|
219
|
-
// word — a 4-token English clause can become 8+ tokens in Korean or
|
|
220
|
-
// Japanese without anything being wrong. Only catch egregious expansion.
|
|
221
|
-
switch (language) {
|
|
222
|
-
case 'en':
|
|
223
|
-
return [1.0, 1.0]; // identity
|
|
224
|
-
case 'he':
|
|
225
|
-
case 'ms':
|
|
226
|
-
return [0.85, 1.5]; // no grammar profile but multi-word translations expand
|
|
227
|
-
case 'es':
|
|
228
|
-
case 'fr':
|
|
229
|
-
case 'de':
|
|
230
|
-
case 'it':
|
|
231
|
-
case 'pt':
|
|
232
|
-
case 'ru':
|
|
233
|
-
case 'uk':
|
|
234
|
-
case 'pl':
|
|
235
|
-
return [0.8, 1.8];
|
|
236
|
-
case 'zh':
|
|
237
|
-
case 'ja':
|
|
238
|
-
case 'ko':
|
|
239
|
-
return [0.65, 2.2]; // CJK adds particles per content word
|
|
240
|
-
case 'tr':
|
|
241
|
-
case 'hi':
|
|
242
|
-
case 'bn':
|
|
243
|
-
case 'qu':
|
|
244
|
-
return [0.7, 2.2]; // SOV with agglutinative markers
|
|
245
|
-
case 'vi':
|
|
246
|
-
case 'th':
|
|
247
|
-
case 'id':
|
|
248
|
-
case 'sw':
|
|
249
|
-
case 'tl':
|
|
250
|
-
return [0.75, 1.8];
|
|
251
|
-
case 'ar':
|
|
252
|
-
return [0.75, 2.0];
|
|
253
|
-
default:
|
|
254
|
-
return [0.7, 2.0];
|
|
255
|
-
}
|
|
256
|
-
}
|
|
257
|
-
|
|
258
|
-
// =============================================================================
|
|
259
|
-
// Aggregate runner
|
|
260
|
-
// =============================================================================
|
|
261
|
-
|
|
262
|
-
export type CheckKind = 'html-parity' | 'literal' | 'truncation';
|
|
263
|
-
|
|
264
|
-
export interface CheckFailure {
|
|
265
|
-
kind: CheckKind;
|
|
266
|
-
error: string;
|
|
267
|
-
}
|
|
268
|
-
|
|
269
|
-
export function runAllChecks(source: string, translated: string, language: string): CheckFailure[] {
|
|
270
|
-
const failures: CheckFailure[] = [];
|
|
271
|
-
|
|
272
|
-
const html = htmlParityCheck(source, translated);
|
|
273
|
-
if (!html.valid) failures.push({ kind: 'html-parity', error: html.error! });
|
|
274
|
-
|
|
275
|
-
const lit = literalPreservationCheck(source, translated);
|
|
276
|
-
if (!lit.valid) failures.push({ kind: 'literal', error: lit.error! });
|
|
277
|
-
|
|
278
|
-
const trunc = tokenCountCheck(source, translated, language);
|
|
279
|
-
if (!trunc.valid) failures.push({ kind: 'truncation', error: trunc.error! });
|
|
280
|
-
|
|
281
|
-
return failures;
|
|
282
|
-
}
|