@lynn123411/dsh-chat-translate 3.0.1 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,64 +0,0 @@
1
- /**
2
- * Mask placeholder tokens — single source of truth for the wire format, the
3
- * matcher and the leftover detector.
4
- *
5
- * Wire format: `⟦xkbdt3⟧` — U+27E6/U+27E7 (mathematical white square brackets)
6
- * around `<4-letter random id><index>`.
7
- *
8
- * Design constraints, each one a measured failure against real MT engines:
9
- *
10
- * - No letters spelling a pronounceable word and no underscores: engines
11
- * abbreviate tokens with such a shape back to their recognizable core, and
12
- * bare `DSH` runs then show in the UI instead of the protected fragments.
13
- * - Random letters keep two mask passes in the same session from colliding and
14
- * make the token unambiguous in the translated text.
15
- * - `matchMaskToken` accepts exactly what this module emits (plus the same
16
- * token with a dropped closing bracket, an engine rewrite seen in practice).
17
- * It deliberately does NOT accept anything else: a translation that damaged
18
- * a token beyond recognition is discarded rather than repaired, because
19
- * repairing it is what put mixed or duplicated text on screen.
20
- */
21
- export declare function randomTokenId(): string;
22
- export interface MaskTokenFormat {
23
- /** Unique per `mask()` call; embedded in every token of that call. */
24
- id: string;
25
- /** Build the placeholder written into the text sent to the translator. */
26
- token: (index: number) => string;
27
- }
28
- export declare function createMaskTokenFormat(): MaskTokenFormat;
29
- /**
30
- * Scanner for complete tokens of this format. A token whose closing bracket the
31
- * engine dropped is still resolved (see `matchMaskToken`), but it is not
32
- * recognized as a complete token here: the space `mask()` inserts sits outside
33
- * that token, and matching a truncated one would strip a space of the
34
- * translation's own.
35
- */
36
- export declare const MASK_TOKEN_PATTERN_SOURCE = "\u27E6([a-z]{4})(\\d+)\u27E7";
37
- /**
38
- * The opening bracket, the id and the index of a token, without its closing
39
- * bracket. Used to resolve a token the engine truncated.
40
- */
41
- export declare const MASK_TOKEN_PREFIX_PATTERN_SOURCE = "\u27E6\\s*([a-z]{4})(\\d+)";
42
- export interface MaskTokenMatch {
43
- /** Index into the mask list of the `mask()` call that produced the token. */
44
- index: number;
45
- /** Length of the matched token in characters. */
46
- length: number;
47
- /** True when the closing bracket was missing and only the prefix matched. */
48
- truncated: boolean;
49
- }
50
- /**
51
- * Match exactly one mask token of this format at `start` in `text`.
52
- *
53
- * The id must match the id of the masking pass that owns the token: every mask
54
- * pass carries its own random id, so a token carrying another id belongs to
55
- * another call and must not be resolved here.
56
- */
57
- export declare function matchMaskToken(text: string, start: number, id: string): MaskTokenMatch | null;
58
- /** Every token of this format in `text`, left to right. */
59
- export declare function findMaskTokens(text: string): Array<{
60
- match: MaskTokenMatch;
61
- raw: string;
62
- }>;
63
- /** True when `text` still carries a token of the current format. */
64
- export declare function hasMaskResidue(text: string): boolean;
@@ -1,45 +0,0 @@
1
- import { hasMaskResidue } from './mask-tokens.ts';
2
- export interface MaskResult {
3
- maskedText: string;
4
- /**
5
- * Restore every protected fragment. Throws `MaskRestoreError` when the
6
- * translated text did not carry the token sequence back intact; the caller
7
- * must discard that translation instead of showing a repaired guess.
8
- */
9
- unmask: (translatedText: string) => string;
10
- }
11
- /** Which spaces `mask()` inserted directly before and after a token. */
12
- export interface InsertedSpacing {
13
- leading: boolean;
14
- trailing: boolean;
15
- }
16
- export declare class MaskRestoreError extends Error {
17
- readonly expectedCount: number;
18
- readonly foundCount: number;
19
- constructor(message: string, expectedCount: number, foundCount: number);
20
- }
21
- export declare class ContentMaskingPipeline {
22
- mask(text: string): MaskResult;
23
- }
24
- /**
25
- * Restore every token of `id` in `text` from `masks`.
26
- *
27
- * Each fragment of the pass is resolved exactly once, no matter how the engine
28
- * reordered the sentence: the index embedded in the token identifies the
29
- * fragment, and the random id identifies the masking pass that owns it. The
30
- * spaces recorded in `inserted` are removed together with their token, so the
31
- * spacing the translation produced around the fragment survives untouched.
32
- *
33
- * @throws MaskRestoreError when the text does not carry every token exactly
34
- * once, or still shows a token of some other pass or a damaged one — the
35
- * caller discards such a translation instead of rendering a repaired guess.
36
- */
37
- export declare function replaceMaskTokens(text: string, id: string, masks: string[], inserted?: InsertedSpacing[]): string;
38
- /**
39
- * True when a translated string still shows a mask token to the user. Callers
40
- * use it to discard a poisoned translation, to keep it out of the caches, and
41
- * to drop poisoned entries when a cache document is loaded (the file on disk
42
- * outlives the process and may carry anything).
43
- */
44
- export declare function isMaskLeak(translatedText: string): boolean;
45
- export { hasMaskResidue };