pi-mega-compact 0.18.0 → 0.18.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/vector-cortex.js +9 -0
- package/dist/config.js +117 -0
- package/dist/extensions/dashboard-server/routes-rag-settings-helpers.js +3 -0
- package/dist/log.js +47 -0
- package/dist/src/config/vector-cortex.js +9 -0
- package/dist/src/config.js +1 -1
- package/dist/src/vector-cortex/encoder/emit-vc2b.js +63 -0
- package/dist/src/vector-cortex/encoder/heads.js +113 -0
- package/dist/src/vector-cortex/encoder/lexical.js +104 -0
- package/dist/src/vector-cortex/encoder/router.js +115 -0
- package/dist/src/vector-cortex/encoder/trigram.js +75 -0
- package/dist/src/vector-cortex/encoder/types.js +53 -0
- package/dist/vector-cortex/encoder/emit-vc2b.js +63 -0
- package/dist/vector-cortex/encoder/heads.js +113 -0
- package/dist/vector-cortex/encoder/lexical.js +104 -0
- package/dist/vector-cortex/encoder/router.js +115 -0
- package/dist/vector-cortex/encoder/trigram.js +75 -0
- package/dist/vector-cortex/encoder/types.js +53 -0
- package/extensions/dashboard-server/routes-rag-settings-helpers.ts +8 -0
- package/package.json +1 -1
- package/src/config/vector-cortex.ts +10 -0
- package/src/config.ts +1 -0
- package/src/vector-cortex/encoder/emit-vc2b.ts +82 -0
- package/src/vector-cortex/encoder/heads.ts +142 -0
- package/src/vector-cortex/encoder/lexical.ts +123 -0
- package/src/vector-cortex/encoder/router.ts +163 -0
- package/src/vector-cortex/encoder/trigram.ts +85 -0
- package/src/vector-cortex/encoder/types.ts +110 -0
|
@@ -73,6 +73,15 @@ export const VC1C_ENABLED = () => sprintFlag("MEGACOMPACT_VC1C");
|
|
|
73
73
|
* encoder runtime's A/B/C selection.
|
|
74
74
|
*/
|
|
75
75
|
export const VC2A_ENABLED = () => sprintFlag("MEGACOMPACT_VC2A");
|
|
76
|
+
/**
|
|
77
|
+
* VC2B — multi-head encoder (VectorSetV1 / HeadCalibrationDraft).
|
|
78
|
+
* Default ON. `MEGACOMPACT_VC2B=0` disables and is byte-identical to the
|
|
79
|
+
* predecessor (the encoder emits no per-head vectors and no fallback-selected
|
|
80
|
+
* event; the trigram/lexical paths themselves are unchanged and are the
|
|
81
|
+
* predecessor's mode-B/C producers). The real consumers are the encoder-heads
|
|
82
|
+
* emit seam and the multi-head encoder producers (heads/trigram/lexical).
|
|
83
|
+
*/
|
|
84
|
+
export const VC2B_ENABLED = () => sprintFlag("MEGACOMPACT_VC2B");
|
|
76
85
|
// ---------------------------------------------------------------------------
|
|
77
86
|
// Breaker state machine constants (TRIAD_RESILIENCE.md §breaker).
|
|
78
87
|
// Rolled numbers for one 60s window; VC0C consumes these at its breaker seam.
|
package/dist/config.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* config.ts — shared default paths/constants for the mega-compact engine.
|
|
3
|
+
*
|
|
4
|
+
* Kept tiny and dependency-free so both the extension entry and unit tests can
|
|
5
|
+
* import it without pulling in pi runtime types.
|
|
6
|
+
*/
|
|
7
|
+
import { join } from "node:path";
|
|
8
|
+
import { homedir } from "node:os";
|
|
9
|
+
/** Default on-disk location for checkpoints + session state. */
|
|
10
|
+
export const STATE_DIR_DEFAULT = join(homedir(), ".pi", "agent", "extensions", "pi-mega-compact");
|
|
11
|
+
/** Pi custom message / entry type used as the dedup sentinel. */
|
|
12
|
+
export const MARKER_TYPE = "mega-compact-marker";
|
|
13
|
+
/**
|
|
14
|
+
* Derive context-window pressure (0–1) from a usage percentage. Used to scale
|
|
15
|
+
* compression strength + keepFrom depth (Fix E): low pct = room to spare,
|
|
16
|
+
* high pct = near the limit. Deterministic; clamps to [0,1].
|
|
17
|
+
*/
|
|
18
|
+
export function pressureFromPct(pct) {
|
|
19
|
+
if (pct == null || Number.isNaN(pct))
|
|
20
|
+
return 0;
|
|
21
|
+
return pct < 0 ? 0 : pct > 100 ? 1 : pct / 100;
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Map pressure → how many recent messages to preserve verbatim. Under low
|
|
25
|
+
* pressure we keep `preserveRecent`; under high pressure we compact deeper,
|
|
26
|
+
* down to `preserveRecentMin`. Never splits a tool pair / anchor floor — the
|
|
27
|
+
* boundary guard (computeDropRange) enforces that downstream.
|
|
28
|
+
*/
|
|
29
|
+
export function preserveRecentForPressure(pressure, preserveRecent, preserveRecentMin) {
|
|
30
|
+
const p = pressure < 0 ? 0 : pressure > 1 ? 1 : pressure;
|
|
31
|
+
const v = Math.round(preserveRecent - (preserveRecent - preserveRecentMin) * p);
|
|
32
|
+
// Floor of 1: even with preserveRecentMin=0 at full pressure, never compact
|
|
33
|
+
// ALL messages — the boundary guard (computeDropRange) needs ≥1 to anchor on.
|
|
34
|
+
return Math.max(1, preserveRecentMin, Math.min(preserveRecent, v));
|
|
35
|
+
}
|
|
36
|
+
/** Clamp a pressure ratio into [0, 1]. */
|
|
37
|
+
function clamp01(p) {
|
|
38
|
+
if (!Number.isFinite(p))
|
|
39
|
+
return 0;
|
|
40
|
+
return p < 0 ? 0 : p > 1 ? 1 : p;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Pressure as a 0–1 ratio from live token usage relative to the compaction
|
|
44
|
+
* threshold. Cheaper + more direct than deriving from a usage percentage when
|
|
45
|
+
* we already have both numbers (the context handler does). Re-exports
|
|
46
|
+
* `pressureFromPct` covers the percentage-only path. (S24.)
|
|
47
|
+
*/
|
|
48
|
+
export function pressureRatio(currentTokens, thresholdTokens) {
|
|
49
|
+
if (!Number.isFinite(currentTokens) || currentTokens <= 0)
|
|
50
|
+
return 0;
|
|
51
|
+
const t = Number.isFinite(thresholdTokens) && thresholdTokens > 0 ? thresholdTokens : 0;
|
|
52
|
+
return clamp01(t > 0 ? currentTokens / t : 0);
|
|
53
|
+
}
|
|
54
|
+
/** Map a 0–1 pressure ratio to a discrete band. (S24.) */
|
|
55
|
+
export function pressureBand(pressure) {
|
|
56
|
+
const p = clamp01(pressure);
|
|
57
|
+
if (p >= 1.0)
|
|
58
|
+
return "mega";
|
|
59
|
+
if (p >= 0.9)
|
|
60
|
+
return "ultra";
|
|
61
|
+
if (p >= 0.75)
|
|
62
|
+
return "high";
|
|
63
|
+
if (p >= 0.5)
|
|
64
|
+
return "medium";
|
|
65
|
+
return "low";
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Memory auto-review cadence (in turns) for a given pressure band. As pressure
|
|
69
|
+
* climbs, the conversation is reviewed more often so durable memories keep pace
|
|
70
|
+
* with the faster context churn. Returns a divisor used as
|
|
71
|
+
* `turn % cadence === 0`. Always >= 1. (S24 — memory cadence tie-in.)
|
|
72
|
+
*/
|
|
73
|
+
export function memoryReviewCadence(band, baseInterval) {
|
|
74
|
+
const base = baseInterval >= 1 ? baseInterval : 1;
|
|
75
|
+
switch (band) {
|
|
76
|
+
case "mega": return Math.max(1, Math.round(base / 5));
|
|
77
|
+
case "ultra": return Math.max(1, Math.round(base / 3));
|
|
78
|
+
case "high": return Math.max(1, Math.round(base / 2));
|
|
79
|
+
case "medium": return Math.max(1, Math.round((base * 2) / 3));
|
|
80
|
+
case "low":
|
|
81
|
+
default: return base;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
// ---------------------------------------------------------------------------
|
|
85
|
+
// S57 RAG Suite feature flags — all default ON with graceful fallback; opt OUT
|
|
86
|
+
// via MEGACOMPACT_<NAME>_DISABLED=true. Every feature degrades to existing
|
|
87
|
+
// behavior on error (non-fatal, best-effort).
|
|
88
|
+
// ---------------------------------------------------------------------------
|
|
89
|
+
function ragFlag(name) {
|
|
90
|
+
const v = process.env[name];
|
|
91
|
+
if (v === undefined)
|
|
92
|
+
return false;
|
|
93
|
+
return v === "true" || v === "1";
|
|
94
|
+
}
|
|
95
|
+
function ragEnabled(name) {
|
|
96
|
+
return !ragFlag(name + "_DISABLED");
|
|
97
|
+
}
|
|
98
|
+
/** B1: Query reformulation (keyword expansion via embedding neighbors). */
|
|
99
|
+
export const RAG_QUERY_REFORMULATION = () => ragEnabled("MEGACOMPACT_QUERY_REFORMULATION");
|
|
100
|
+
/** B2: Tiered recall router (L0 cache → L1 FTS5 → L2 HNSW). */
|
|
101
|
+
export const RAG_TIERED_ROUTER = () => ragEnabled("MEGACOMPACT_TIERED_ROUTER");
|
|
102
|
+
/** B3: Recall quality metrics (precision/recall scoring + logging). */
|
|
103
|
+
export const RAG_RECALL_METRICS = () => ragEnabled("MEGACOMPACT_RECALL_METRICS");
|
|
104
|
+
/** B4: Memory graph traversal (dashboard-oriented). */
|
|
105
|
+
export const RAG_MEMORY_GRAPH = () => ragEnabled("MEGACOMPACT_MEMORY_GRAPH");
|
|
106
|
+
/** B5: HyDE — generate a hypothetical answer doc via LLM. Auto-ON when an
|
|
107
|
+
* HttpEmbedder is active (the LLM is configured for indexing); opt OUT with
|
|
108
|
+
* MEGACOMPACT_HYDE_DISABLED=true. TrigramEmbedder path is unaffected. */
|
|
109
|
+
export const RAG_HYDE_ENABLED = () => ragEnabled("MEGACOMPACT_HYDE");
|
|
110
|
+
/** Spec 1: vbrainstorm visual design migration for the dashboard. */
|
|
111
|
+
export const NEW_UI = () => ragEnabled("MEGACOMPACT_NEW_UI");
|
|
112
|
+
// ---------------------------------------------------------------------------
|
|
113
|
+
// Vector-cortex flags + breaker constants (VC0A+). Positive sprint flags,
|
|
114
|
+
// default ON, `=0`/`_DISABLED` off. Re-exported from src/config/vector-cortex.ts
|
|
115
|
+
// so root consumers share one source of truth.
|
|
116
|
+
// ---------------------------------------------------------------------------
|
|
117
|
+
export { VC0A_ENABLED, VC0B_ENABLED, VC1A_ENABLED, VC0C_ENABLED, VC1B_ENABLED, VC1C_ENABLED, VC2A_ENABLED, VC2B_ENABLED, BREAKER_WINDOW_MS, BREAKER_MIN_ATTEMPTS, BREAKER_PERF_FAILURES, BREAKER_PERF_FAILURE_RATE, BREAKER_CORRECTNESS_FAILURES, BREAKER_COOLDOWN_MS, BREAKER_PROBE_COUNT, BREAKER_RETRY_BASE_MS, BREAKER_RETRY_CAP_MS, BREAKER_RETRY_JITTER, BREAKER_HYSTERESIS_FAILURE_RATE, BREAKER_HYSTERESIS_BUDGET_P95_MS, BREAKER_MIN_HEALTHY_RESIDENCE_MS, } from "./config/vector-cortex.js";
|
|
@@ -127,6 +127,8 @@ export const SETTINGS = [
|
|
|
127
127
|
str("MEGACOMPACT_RAPTOR_MODEL", "RAPTOR Summary Model", "Ollama model for cluster summarization (empty = extractive)", ""),
|
|
128
128
|
str("MEGACOMPACT_RAPTOR_URL", "RAPTOR Ollama URL", "Ollama endpoint for RAPTOR summarization", "http://127.0.0.1:11434"),
|
|
129
129
|
num("MEGACOMPACT_EMBED_CACHE", "Embed Cache Size", "Embedding cache entries (0 = disabled)", 256, 0, 10000),
|
|
130
|
+
num("MEGACOMPACT_EMBEDDING_BATCH_TOKENS", "Embedding Batch Tokens", "Oversized-prompt chunking limit (tokens) for the BYO localhost embedder; text above this is chunked + mean-pooled", 2048, 64, 8192),
|
|
131
|
+
num("MEGACOMPACT_EMBEDDING_CHARS_PER_TOKEN", "Embedding Chars per Token", "Estimated characters per token used for embedder chunking size", 4, 1, 32),
|
|
130
132
|
],
|
|
131
133
|
},
|
|
132
134
|
{
|
|
@@ -139,6 +141,7 @@ export const SETTINGS = [
|
|
|
139
141
|
boolDirect("MEGACOMPACT_VC0C", "VC0C Live Safety Envelope", "TriadResult/Breaker live circuit breaker (60s window, 20 attempts, 30s cooldown, 3 probes, 5min healthy residence) + durable spool before provider invocation; manual reset clears cooldown but never evidence. OFF = mode C, unchanged transcript, byte-identical.", true),
|
|
140
142
|
boolDirect("MEGACOMPACT_VC1C", "VC1C Cross-Language Conformance v2", "FixtureManifestV2 canonical manifest validator + DowngradeReport deterministic downgrade export + MinHashV2 exact big-integer signatures and the M4 copy/validate/switch minhash-v2 migration (seed table frozen, cross-language byte-exact). OFF = mode C, v1 sync dedup scan unchanged, byte-identical.", true),
|
|
141
143
|
boolDirect("MEGACOMPACT_VC2A", "VC2A Offline Model Runtime", "ModelManifestV1 digest-before-load ONNX runtime (opset17/batch1/max512) + asset-free trigram demotion. Asset path assets/vector-cortex/encoder-v1 is immutable/digest-pinned. OFF = mode C, byte-identical to predecessor.", true),
|
|
144
|
+
boolDirect("MEGACOMPACT_VC2B", "VC2B Multi-Head Encoder", "VectorSetV1 five L2-normalized heads (384/128/128/64/32) with head-calibration draft + asset-free trigram B (512d) and lexical C fallbacks, plus the per-head emit seam. OFF = mode C, no per-head vectors emitted, byte-identical predecessor.", true),
|
|
142
145
|
],
|
|
143
146
|
},
|
|
144
147
|
{
|
package/dist/log.js
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* log.ts — tiny append-only structured logger.
|
|
3
|
+
*
|
|
4
|
+
* Writes one JSON object per line to a log file (default:
|
|
5
|
+
* ~/.pi/agent/extensions/mega-compact.log). Best-effort: logging never throws
|
|
6
|
+
* into the extension. Pi-agnostic and dependency-free so it can be unit-tested.
|
|
7
|
+
*/
|
|
8
|
+
import { appendFileSync, mkdirSync } from "node:fs";
|
|
9
|
+
import { dirname, join } from "node:path";
|
|
10
|
+
import { STATE_DIR_DEFAULT } from "./config.js";
|
|
11
|
+
/** Default log path lives alongside the state dir. */
|
|
12
|
+
export function defaultLogPath() {
|
|
13
|
+
return join(STATE_DIR_DEFAULT, "mega-compact.log");
|
|
14
|
+
}
|
|
15
|
+
export class Logger {
|
|
16
|
+
path;
|
|
17
|
+
enabled;
|
|
18
|
+
/** Monotonic clock injected by the caller so the module stays deterministic. */
|
|
19
|
+
now;
|
|
20
|
+
constructor(opts = {}) {
|
|
21
|
+
this.path = opts.path ?? defaultLogPath();
|
|
22
|
+
this.enabled = opts.enabled ?? true;
|
|
23
|
+
this.now = opts.now ?? (() => Date.now());
|
|
24
|
+
}
|
|
25
|
+
/** Append one structured line. Swallows all I/O errors. */
|
|
26
|
+
log(level, event, fields = {}) {
|
|
27
|
+
if (!this.enabled)
|
|
28
|
+
return;
|
|
29
|
+
const entry = { ts: this.now(), level, event, ...fields };
|
|
30
|
+
try {
|
|
31
|
+
mkdirSync(dirname(this.path), { recursive: true });
|
|
32
|
+
appendFileSync(this.path, `${JSON.stringify(entry)}\n`);
|
|
33
|
+
}
|
|
34
|
+
catch {
|
|
35
|
+
/* best-effort: never break the extension on a log failure */
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
info(event, fields) {
|
|
39
|
+
this.log("info", event, fields);
|
|
40
|
+
}
|
|
41
|
+
warn(event, fields) {
|
|
42
|
+
this.log("warn", event, fields);
|
|
43
|
+
}
|
|
44
|
+
error(event, fields) {
|
|
45
|
+
this.log("error", event, fields);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
@@ -73,6 +73,15 @@ export const VC1C_ENABLED = () => sprintFlag("MEGACOMPACT_VC1C");
|
|
|
73
73
|
* encoder runtime's A/B/C selection.
|
|
74
74
|
*/
|
|
75
75
|
export const VC2A_ENABLED = () => sprintFlag("MEGACOMPACT_VC2A");
|
|
76
|
+
/**
|
|
77
|
+
* VC2B — multi-head encoder (VectorSetV1 / HeadCalibrationDraft).
|
|
78
|
+
* Default ON. `MEGACOMPACT_VC2B=0` disables and is byte-identical to the
|
|
79
|
+
* predecessor (the encoder emits no per-head vectors and no fallback-selected
|
|
80
|
+
* event; the trigram/lexical paths themselves are unchanged and are the
|
|
81
|
+
* predecessor's mode-B/C producers). The real consumers are the encoder-heads
|
|
82
|
+
* emit seam and the multi-head encoder producers (heads/trigram/lexical).
|
|
83
|
+
*/
|
|
84
|
+
export const VC2B_ENABLED = () => sprintFlag("MEGACOMPACT_VC2B");
|
|
76
85
|
// ---------------------------------------------------------------------------
|
|
77
86
|
// Breaker state machine constants (TRIAD_RESILIENCE.md §breaker).
|
|
78
87
|
// Rolled numbers for one 60s window; VC0C consumes these at its breaker seam.
|
package/dist/src/config.js
CHANGED
|
@@ -114,4 +114,4 @@ export const NEW_UI = () => ragEnabled("MEGACOMPACT_NEW_UI");
|
|
|
114
114
|
// default ON, `=0`/`_DISABLED` off. Re-exported from src/config/vector-cortex.ts
|
|
115
115
|
// so root consumers share one source of truth.
|
|
116
116
|
// ---------------------------------------------------------------------------
|
|
117
|
-
export { VC0A_ENABLED, VC0B_ENABLED, VC1A_ENABLED, VC0C_ENABLED, VC1B_ENABLED, VC1C_ENABLED, VC2A_ENABLED, BREAKER_WINDOW_MS, BREAKER_MIN_ATTEMPTS, BREAKER_PERF_FAILURES, BREAKER_PERF_FAILURE_RATE, BREAKER_CORRECTNESS_FAILURES, BREAKER_COOLDOWN_MS, BREAKER_PROBE_COUNT, BREAKER_RETRY_BASE_MS, BREAKER_RETRY_CAP_MS, BREAKER_RETRY_JITTER, BREAKER_HYSTERESIS_FAILURE_RATE, BREAKER_HYSTERESIS_BUDGET_P95_MS, BREAKER_MIN_HEALTHY_RESIDENCE_MS, } from "./config/vector-cortex.js";
|
|
117
|
+
export { VC0A_ENABLED, VC0B_ENABLED, VC1A_ENABLED, VC0C_ENABLED, VC1B_ENABLED, VC1C_ENABLED, VC2A_ENABLED, VC2B_ENABLED, BREAKER_WINDOW_MS, BREAKER_MIN_ATTEMPTS, BREAKER_PERF_FAILURES, BREAKER_PERF_FAILURE_RATE, BREAKER_CORRECTNESS_FAILURES, BREAKER_COOLDOWN_MS, BREAKER_PROBE_COUNT, BREAKER_RETRY_BASE_MS, BREAKER_RETRY_CAP_MS, BREAKER_RETRY_JITTER, BREAKER_HYSTERESIS_FAILURE_RATE, BREAKER_HYSTERESIS_BUDGET_P95_MS, BREAKER_MIN_HEALTHY_RESIDENCE_MS, } from "./config/vector-cortex.js";
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vector-cortex/encoder/emit-vc2b.ts — VC2B observability seam.
|
|
3
|
+
*
|
|
4
|
+
* Owns the two VC2B events (task 5), gated on `MEGACOMPACT_VC2B` so the
|
|
5
|
+
* flag-OFF path emits zero events (mode C parity, byte-identical predecessor):
|
|
6
|
+
*
|
|
7
|
+
* vector_cortex_encoder_heads_emitted — a multi-head VectorSetV1 produced
|
|
8
|
+
* vector_cortex_encoder_fallback_selected — a mode B/C fallback selected
|
|
9
|
+
*
|
|
10
|
+
* No dashboard or API change is necessary for this internal sprint (task 5).
|
|
11
|
+
* Every event is a JSON line with `ts` + `event` (ENGINEERING_PRACTICES §8); the
|
|
12
|
+
* emitters are non-fatal (never break the agent loop). Pi-agnostic, zero network
|
|
13
|
+
* (PREVENT-PI-004), no `any` (PREVENT-011).
|
|
14
|
+
*/
|
|
15
|
+
import { VC2B_ENABLED } from "../../config/vector-cortex.js";
|
|
16
|
+
import { Logger } from "../../log.js";
|
|
17
|
+
/** A flag-gated no-op reporter (zero emissions, structural no-op). */
|
|
18
|
+
export const NOOP_VC2B_REPORTER = {
|
|
19
|
+
headsEmitted: () => { },
|
|
20
|
+
fallbackSelected: () => { },
|
|
21
|
+
};
|
|
22
|
+
/**
|
|
23
|
+
* The default emitter: routes both VC2B events into the append-only structured
|
|
24
|
+
* logger (`src/log.ts`) as JSON lines with `ts` + `event`. Supplying `emit:` to
|
|
25
|
+
* `createEncoderHeadsReporter` replaces this with a caller-provided sink (used
|
|
26
|
+
* by tests and downstream consumers). Making the default a REAL producer means a
|
|
27
|
+
* caller that just invokes the producer seam (`encodeOrFallback`, `encodeVectorSet`,
|
|
28
|
+
* `selectTrigramBFallback`, `selectLexicalC`) without injecting an emitter still
|
|
29
|
+
* yields structured telemetry instead of silently dropping every event (task 5,
|
|
30
|
+
* code-review Q01). Best-effort: the logger swallows all I/O errors.
|
|
31
|
+
*/
|
|
32
|
+
function defaultEmitFor(logPath) {
|
|
33
|
+
const logger = new Logger(logPath === undefined ? {} : { path: logPath });
|
|
34
|
+
return (event, fields) => {
|
|
35
|
+
logger.info(event, fields);
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Flag-gated emit, defaulting to a real logger-backed sink. The returned
|
|
40
|
+
* reporter is itself flag-gated (`VC2B_ENABLED`), so wiring it into a producer
|
|
41
|
+
* seam yields zero emissions when `MEGACOMPACT_VC2B=0` (byte-identical to the
|
|
42
|
+
* predecessor). Pass an explicit `emit` to route elsewhere (tests, downstream
|
|
43
|
+
* consumers); omit it to emit real structured log lines (Q01: the default is a
|
|
44
|
+
* live producer, not a silent no-op). `opts.logPath` only redirects the default
|
|
45
|
+
* sink and is ignored when `emit` is supplied.
|
|
46
|
+
*/
|
|
47
|
+
export function createEncoderHeadsReporter(emit, opts = {}) {
|
|
48
|
+
const sink = emit ?? defaultEmitFor(opts.logPath);
|
|
49
|
+
const fire = (event, fields) => {
|
|
50
|
+
if (!VC2B_ENABLED())
|
|
51
|
+
return;
|
|
52
|
+
try {
|
|
53
|
+
sink(event, { ...fields, ts: new Date().toISOString() });
|
|
54
|
+
}
|
|
55
|
+
catch {
|
|
56
|
+
/* non-fatal observability */
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
return {
|
|
60
|
+
headsEmitted: (fields) => fire("vector_cortex_encoder_heads_emitted", fields),
|
|
61
|
+
fallbackSelected: (fields) => fire("vector_cortex_encoder_fallback_selected", fields),
|
|
62
|
+
};
|
|
63
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vector-cortex/encoder/heads.ts — VC2B multi-head encoder (tasks 1–2).
|
|
3
|
+
*
|
|
4
|
+
* Produces a `VectorSetV1`: five independent L2-normalized projection heads in
|
|
5
|
+
* STABLE order — semantic 384, dependency 128, contradiction 128, cacheStability
|
|
6
|
+
* 64, payloadRouting 32 (MODEL_ASSET §decision record). Each head L2-normalizes
|
|
7
|
+
* its raw projection; a zero-norm projection maps to an all-zero vector (task 2).
|
|
8
|
+
*
|
|
9
|
+
* The raw per-head projection is a deterministic seeded compression of the input
|
|
10
|
+
* token sequence (seeded by `ENCODER_SEED` and the head's stable index), which
|
|
11
|
+
* mirrors the VC2A `projectSemantic` placeholder pattern: the contract, shape
|
|
12
|
+
* gating, normalization, zero-norm mapping, ordering and loss/seed constants are
|
|
13
|
+
* all normative here; real trained weights are substituted in VC2C. This keeps
|
|
14
|
+
* the mode-A multi-head path fully testable end-to-end today with zero network.
|
|
15
|
+
*
|
|
16
|
+
* The VC2B emit seam (task 5) is wired: producing a VectorSetV1 emits
|
|
17
|
+
* `vector_cortex_encoder_heads_emitted`; selecting a mode B/C fallback emits
|
|
18
|
+
* `vector_cortex_encoder_fallback_selected` — both gated on MEGACOMPACT_VC2B.
|
|
19
|
+
*
|
|
20
|
+
* Pi-agnostic, zero network (PREVENT-PI-004), no `any` (PREVENT-011).
|
|
21
|
+
*/
|
|
22
|
+
import { ENCODER_HEAD_DIMS, ENCODER_HEAD_ORDER, ENCODER_HEAD_LOSS_WEIGHTS, ENCODER_HEAD_LOSS_SUM, ENCODER_SEED, } from "./types.js";
|
|
23
|
+
import { createEncoderHeadsReporter, NOOP_VC2B_REPORTER, } from "./emit-vc2b.js";
|
|
24
|
+
/** The stable head index of a head name (its position in ENCODER_HEAD_ORDER). */
|
|
25
|
+
const HEAD_INDEX = {
|
|
26
|
+
semantic: 0,
|
|
27
|
+
dependency: 1,
|
|
28
|
+
contradiction: 2,
|
|
29
|
+
cacheStability: 3,
|
|
30
|
+
payloadRouting: 4,
|
|
31
|
+
};
|
|
32
|
+
/** Deterministic 32-bit LCG step (matches runtime.ts projectSemantic). */
|
|
33
|
+
function nextState(state) {
|
|
34
|
+
return (state * 1664525 + 1013904223) >>> 0;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* L2-normalize a float vector in place semantics (returns a new Float32Array).
|
|
38
|
+
* A zero-norm (or empty) input maps to an all-zero vector of the same length
|
|
39
|
+
* (task 2: "mapping zero norm to an all-zero vector"). All finite.
|
|
40
|
+
*/
|
|
41
|
+
export function l2Normalize(values) {
|
|
42
|
+
const out = new Float32Array(values.length);
|
|
43
|
+
let sum = 0;
|
|
44
|
+
for (const v of values)
|
|
45
|
+
sum += v * v;
|
|
46
|
+
const norm = Math.sqrt(sum);
|
|
47
|
+
if (!(norm > 0))
|
|
48
|
+
return out; // zero norm -> all-zero
|
|
49
|
+
for (let i = 0; i < values.length; i++)
|
|
50
|
+
out[i] = values[i] / norm;
|
|
51
|
+
return out;
|
|
52
|
+
}
|
|
53
|
+
/** L2 norm of a Float32Array (0 for empty/all-zero). */
|
|
54
|
+
export function l2Norm(values) {
|
|
55
|
+
let sum = 0;
|
|
56
|
+
for (const v of values)
|
|
57
|
+
sum += v * v;
|
|
58
|
+
return Math.sqrt(sum);
|
|
59
|
+
}
|
|
60
|
+
/** A deterministic per-head projection over the token sequence, pre-normalization. */
|
|
61
|
+
function projectRaw(head, tokens, seed) {
|
|
62
|
+
const dim = ENCODER_HEAD_DIMS[head];
|
|
63
|
+
const out = new Float32Array(dim);
|
|
64
|
+
// An EMPTY token sequence has no signal: the raw projection is the zero vector,
|
|
65
|
+
// so after L2 normalization it maps to the all-zero vector (task 2: "mapping
|
|
66
|
+
// zero norm to an all-zero vector"; ENC-ZERO-002). This keeps empty input
|
|
67
|
+
// finite and zero-norm instead of seeding spurious unit-norm noise.
|
|
68
|
+
if (tokens.length === 0)
|
|
69
|
+
return out;
|
|
70
|
+
// Mix the stable head index + ENCODER_SEED + seed into a per-head state so
|
|
71
|
+
// each head is a distinct independent projection (failure-triad independence).
|
|
72
|
+
let state = (((ENCODER_SEED ^ HEAD_INDEX[head]) >>> 0) ^ (seed >>> 0)) ^ 0x9e3779b9;
|
|
73
|
+
for (const t of tokens)
|
|
74
|
+
state = nextState(state ^ ((t >>> 0) * 2654435761));
|
|
75
|
+
for (let i = 0; i < dim; i++) {
|
|
76
|
+
state = nextState(state ^ seed);
|
|
77
|
+
out[i] = (state / 4294967296) * 2 - 1;
|
|
78
|
+
}
|
|
79
|
+
return out;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Compute one head's L2-normalized vector (all-zero on zero norm) for a token
|
|
83
|
+
* sequence. Deterministic for a given seed (repeat drift == 0).
|
|
84
|
+
*/
|
|
85
|
+
export function projectHead(head, tokens, seed = ENCODER_SEED) {
|
|
86
|
+
const raw = projectRaw(head, tokens, seed);
|
|
87
|
+
return { head, dim: ENCODER_HEAD_DIMS[head], values: l2Normalize(raw) };
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Encode a token sequence into a `VectorSetV1`: the five heads in stable order,
|
|
91
|
+
* each L2-normalized (all-zero on zero norm). Emits `heads_emitted` via the
|
|
92
|
+
* reporter (non-fatal, flag-gated). Deterministic for a given seed.
|
|
93
|
+
*/
|
|
94
|
+
export function encodeVectorSet(tokens, options = {}) {
|
|
95
|
+
const seed = options.seed ?? ENCODER_SEED;
|
|
96
|
+
const reporter = options.reporter ?? createEncoderHeadsReporter();
|
|
97
|
+
const heads = ENCODER_HEAD_ORDER.map((h) => projectHead(h, tokens, seed));
|
|
98
|
+
reporter.headsEmitted({
|
|
99
|
+
heads: heads.length,
|
|
100
|
+
dims: heads.map((h) => h.dim).join("/"),
|
|
101
|
+
normalized: true,
|
|
102
|
+
tokens: tokens.length,
|
|
103
|
+
});
|
|
104
|
+
return { schema: "vector-set-v1", inputTokens: [...tokens], heads, normalized: true };
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* The per-head loss weights (must sum to ENCODER_HEAD_LOSS_SUM exactly).
|
|
108
|
+
* Exposed for training/tests to assert the normative .35/.20/.20/.15/.10 split.
|
|
109
|
+
*/
|
|
110
|
+
export function headLossWeights() {
|
|
111
|
+
return { ...ENCODER_HEAD_LOSS_WEIGHTS };
|
|
112
|
+
}
|
|
113
|
+
export { ENCODER_HEAD_ORDER, ENCODER_HEAD_DIMS, ENCODER_HEAD_LOSS_SUM, ENCODER_SEED, NOOP_VC2B_REPORTER };
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vector-cortex/encoder/lexical.ts — VC2B mode C: token/phrase lexical encoder.
|
|
3
|
+
*
|
|
4
|
+
* Lexical C is a token/phrase lexical feature generator used when both mode A
|
|
5
|
+
* (learned asset) and mode B (trigram) are unavailable or fail. It is continuity,
|
|
6
|
+
* NOT semantic completeness: C operates on exact current tokens/phrases only and
|
|
7
|
+
* MUST state that it has lost old semantic context (task 4 + TRIAD_RESILIENCE.
|
|
8
|
+
* "C is continuity, not semantic completeness: it may omit old context and must
|
|
9
|
+
* report that limitation").
|
|
10
|
+
*
|
|
11
|
+
* C never imports the learned asset or learned calibration (task 4): it is a
|
|
12
|
+
* pure token/phrase lexical projection (token counts + phrase hashes) computed
|
|
13
|
+
* from the exact input. It is independently implemented from B (which hashes
|
|
14
|
+
* byte-level trigrams) — C works at the token/phrase level, B at the byte-ngram
|
|
15
|
+
* level, so the two share no algorithm.
|
|
16
|
+
*
|
|
17
|
+
* Authority outage freezes derived high-water: C never advances any derived
|
|
18
|
+
* frontier; it is purely a local reconstruction from the exact present tokens.
|
|
19
|
+
*
|
|
20
|
+
* Pi-agnostic, zero network (PREVENT-PI-004), no `any` (PREVENT-011).
|
|
21
|
+
*/
|
|
22
|
+
import { createHash } from "node:crypto";
|
|
23
|
+
import { l2Normalize } from "./heads.js";
|
|
24
|
+
import { createEncoderHeadsReporter, } from "./emit-vc2b.js";
|
|
25
|
+
/** Fixed output width of lexical C. */
|
|
26
|
+
export const ENCODER_LEXICAL_WIDTH = 256;
|
|
27
|
+
/** The documented limitation lexical C reports (continuity, not semantics). */
|
|
28
|
+
export const ENCODER_LEXICAL_LIMITATION = "lexical C: token/phrase-level continuity only; old semantic context is omitted";
|
|
29
|
+
export function tokenizeLexical(text) {
|
|
30
|
+
// Split into lowercase token/phrase units on non-alphanumeric boundaries.
|
|
31
|
+
return text.toLowerCase().split(/[^a-z0-9_]+/).filter((t) => t.length > 0);
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Normalize a single token/phrase unit the same way `tokenizeLexical` does for
|
|
35
|
+
* the string form: lowercase and strip leading/trailing non-alphanumeric runs.
|
|
36
|
+
* Applied to array-form tokens so both accepted input forms of `embedLexical`
|
|
37
|
+
* hash identical conceptual content to identical buckets, regardless of how the
|
|
38
|
+
* caller chose to pass it (code-review Q03).
|
|
39
|
+
*/
|
|
40
|
+
function normalizeToken(tok) {
|
|
41
|
+
return tok.toLowerCase().split(/[^a-z0-9_]+/).join("");
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Resolve either accepted input form to a normalized token sequence: the string
|
|
45
|
+
* form routes through `tokenizeLexical` (split on non-alphanumeric boundaries,
|
|
46
|
+
* fragmented tokens discarded); the array form applies the same per-token
|
|
47
|
+
* lowercase/strip normalization and drops tokens that normalize to empty. Both
|
|
48
|
+
* paths therefore agree on hash buckets for the same conceptual content.
|
|
49
|
+
*/
|
|
50
|
+
function resolveTokens(tokensOrText) {
|
|
51
|
+
if (typeof tokensOrText === "string")
|
|
52
|
+
return tokenizeLexical(tokensOrText);
|
|
53
|
+
const out = [];
|
|
54
|
+
for (const raw of tokensOrText) {
|
|
55
|
+
const norm = normalizeToken(raw);
|
|
56
|
+
if (norm.length > 0)
|
|
57
|
+
out.push(norm);
|
|
58
|
+
}
|
|
59
|
+
return out;
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Encode a token/phrase sequence into a 256-dim L2-normalized lexical vector
|
|
63
|
+
* (all-zero on empty input). Features: exact token count + token id-hash sums +
|
|
64
|
+
* phrase-adjacency hashes. Deterministic (repeat drift == 0). Pure local compute.
|
|
65
|
+
*/
|
|
66
|
+
export function embedLexical(tokensOrText) {
|
|
67
|
+
const width = ENCODER_LEXICAL_WIDTH;
|
|
68
|
+
const out = new Float32Array(width);
|
|
69
|
+
const tokens = resolveTokens(tokensOrText);
|
|
70
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
71
|
+
const tok = tokens[i];
|
|
72
|
+
const h = createHash("sha256").update(`t:${tok}`).digest();
|
|
73
|
+
const bucket = h.readUInt32BE(0) % width;
|
|
74
|
+
const weight = (h.readUInt32BE(4) / 4294967295) * 2 - 1;
|
|
75
|
+
out[bucket] += weight;
|
|
76
|
+
// Phrase adjacency: bigram hash blended in so semantic-free ordering matters.
|
|
77
|
+
if (i > 0) {
|
|
78
|
+
const pair = createHash("sha256").update(`p:${tokens[i - 1]}:${tok}`).digest();
|
|
79
|
+
const pb = pair.readUInt32BE(0) % width;
|
|
80
|
+
const pw = (pair.readUInt32BE(4) / 4294967295) * 2 - 1;
|
|
81
|
+
out[pb] += pw;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
return l2Normalize(out);
|
|
85
|
+
}
|
|
86
|
+
/** The documented limitation string, surfaced when lexical C is selected. */
|
|
87
|
+
export function selectLexicalC(options = {}) {
|
|
88
|
+
const reporter = options.reporter ?? createEncoderHeadsReporter();
|
|
89
|
+
const selection = {
|
|
90
|
+
ok: true,
|
|
91
|
+
mode: "C",
|
|
92
|
+
dim: ENCODER_LEXICAL_WIDTH,
|
|
93
|
+
width: ENCODER_LEXICAL_WIDTH,
|
|
94
|
+
limitation: ENCODER_LEXICAL_LIMITATION,
|
|
95
|
+
};
|
|
96
|
+
reporter.fallbackSelected({
|
|
97
|
+
mode: selection.mode,
|
|
98
|
+
dim: selection.dim,
|
|
99
|
+
width: selection.width,
|
|
100
|
+
limitation: selection.limitation,
|
|
101
|
+
});
|
|
102
|
+
return selection;
|
|
103
|
+
}
|
|
104
|
+
export { l2Normalize };
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vector-cortex/encoder/router.ts — VC2B encode-or-fallback router (S2).
|
|
3
|
+
*
|
|
4
|
+
* The single production seam that catches a real VC2A `load()` failure and
|
|
5
|
+
* hands off to the independently initialized VC2B fallbacks:
|
|
6
|
+
*
|
|
7
|
+
* - mode A: the learned multi-head projection — an EncoderRuntime that
|
|
8
|
+
* verifies+loads a local qualified ONNX (VC2A) plus `encodeVectorSet`
|
|
9
|
+
* producing VectorSetV1, emitting `vector_cortex_encoder_heads_emitted`;
|
|
10
|
+
* - when that `load()` fails for ANY reason (removed model ->
|
|
11
|
+
* ENC_ASSET_UNREADABLE, digest mismatch -> ENC_DIGEST_MISMATCH, missing
|
|
12
|
+
* manifest -> ENC_MANIFEST_INVALID, unsupported platform, RSS over budget,
|
|
13
|
+
* ...) the router catches the real failure and selects the independently
|
|
14
|
+
* initialized asset-free trigram B (`selectTrigramBFallback`) — or lexical C
|
|
15
|
+
* when the runtime reports B-unavailable / the caller forces mode C. The
|
|
16
|
+
* `vector_cortex_encoder_fallback_selected` event fires from the REAL
|
|
17
|
+
* producer seam, not test wiring (task 5 + code-review S1).
|
|
18
|
+
*
|
|
19
|
+
* Best-effort and non-fatal: every branch returns an explicit verdict and never
|
|
20
|
+
* throws across the boundary into the agent loop. Flag-OFF parity: with
|
|
21
|
+
* `MEGACOMPACT_VC2B=0` the VC2B reporter is a no-op (zero emissions), and the
|
|
22
|
+
* mode-A path is governed by the VC2A runtime itself — the router only adds the
|
|
23
|
+
* fallback handoff and changes no producer bytes.
|
|
24
|
+
*
|
|
25
|
+
* Pi-agnostic, zero network (PREVENT-PI-004), no `any` (PREVENT-011).
|
|
26
|
+
*/
|
|
27
|
+
import { createEncoderRuntime } from "./runtime.js";
|
|
28
|
+
import { encodeVectorSet } from "./heads.js";
|
|
29
|
+
import { embedTrigram512, selectTrigramBFallback } from "./trigram.js";
|
|
30
|
+
import { embedLexical, selectLexicalC } from "./lexical.js";
|
|
31
|
+
import { createEncoderHeadsReporter } from "./emit-vc2b.js";
|
|
32
|
+
import { ENC_FAIL, } from "./types.js";
|
|
33
|
+
/** Deterministic text derived from an int token sequence so the asset-free
|
|
34
|
+
* fallback producers operate on the same authority the learned path encoded. */
|
|
35
|
+
function textFromTokens(tokens) {
|
|
36
|
+
return tokens.join("-");
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Try to produce an encoding for an input token sequence. Mode A: verify+load
|
|
40
|
+
* the local learned asset via the EncoderRuntime; on a real `load()` failure
|
|
41
|
+
* (removed model, digest mismatch, missing manifest, ...) the router catches it
|
|
42
|
+
* and selects the independently initialized trigram B or lexical C, emitting
|
|
43
|
+
* `vector_cortex_encoder_fallback_selected` (and `heads_emitted` when A wins).
|
|
44
|
+
*/
|
|
45
|
+
export function encodeOrFallback(input, assetDir, options = {}) {
|
|
46
|
+
const reporter = options.reporter ?? createEncoderHeadsReporter();
|
|
47
|
+
const tokens = Array.isArray(input?.tokens) ? input.tokens : [];
|
|
48
|
+
const runtime = options.runtime ?? createEncoderRuntime();
|
|
49
|
+
// A caller that explicitly forces a fallback mode wins even over the empty-
|
|
50
|
+
// input degenerate case: the forced-mode contract must hold for ANY input, so
|
|
51
|
+
// handle forceFallback before the empty-input selection below (Q02). A forced
|
|
52
|
+
// B/C is NOT a rollback and NOT a demotion — it is an intentional selection
|
|
53
|
+
// for capacity/testing — so the verdict carries `code: null`, never
|
|
54
|
+
// ENC_FAIL.ROLLBACK, so a consumer that reads `verdict.code` as "what
|
|
55
|
+
// triggered the fallback" won't misread a deliberately forced mode as a
|
|
56
|
+
// rollback and take rollback-specific action (code-review Q04).
|
|
57
|
+
if (options.forceFallback !== undefined) {
|
|
58
|
+
if (options.forceFallback === "C") {
|
|
59
|
+
const sel = selectLexicalC({ reporter });
|
|
60
|
+
const vector = embedLexical(textFromTokens(tokens));
|
|
61
|
+
return { ok: true, mode: "C", vector, width: sel.width, limitation: sel.limitation, code: null };
|
|
62
|
+
}
|
|
63
|
+
const sel = selectTrigramBFallback({ reporter });
|
|
64
|
+
const vector = embedTrigram512(textFromTokens(tokens));
|
|
65
|
+
return { ok: true, mode: "B", vector, width: sel.width, limitation: null, code: null };
|
|
66
|
+
}
|
|
67
|
+
// Empty input yields the asset-free fallback (finite, deterministic zero
|
|
68
|
+
// vector, ENC-ZERO-002). This is legitimate degenerate behavior, NOT a shape
|
|
69
|
+
// failure, so the verdict carries no failure code (code === null) — a consumer
|
|
70
|
+
// that interprets `code` as "what went wrong" must not misread a valid all-zero
|
|
71
|
+
// B vector as a shape rejection (Q05). Only reached when no mode is forced, so
|
|
72
|
+
// a forced B/C is never subverted by empty tokens.
|
|
73
|
+
if (tokens.length === 0) {
|
|
74
|
+
const sel = selectTrigramBFallback({ reporter });
|
|
75
|
+
const vector = embedTrigram512("");
|
|
76
|
+
return { ok: true, mode: "B", vector, width: sel.width, limitation: null, code: null };
|
|
77
|
+
}
|
|
78
|
+
// Mode A attempt — reached only when no mode is forced (the forceFallback
|
|
79
|
+
// guard at the top already returned), so `options.forceFallback` is undefined here.
|
|
80
|
+
const loaded = runtime.load(assetDir);
|
|
81
|
+
if (!loaded.ok) {
|
|
82
|
+
return fallbackFromLoad(loaded, reporter, tokens);
|
|
83
|
+
}
|
|
84
|
+
// Q01/Q03: a qualified mode-A load is not enough — the verified per-manifest
|
|
85
|
+
// token capacity (maxTokens, <= global 512) must also be enforced before we
|
|
86
|
+
// produce a VectorSetV1. run inference over the input; an over-cap sequence
|
|
87
|
+
// (e.g. 100 tokens against a verified maxTokens=64 manifest) or an
|
|
88
|
+
// over-budget inference is rejected here and routed to the B/C fallback with
|
|
89
|
+
// the real failure code, rather than silently emitting an ok:true mode-A
|
|
90
|
+
// set whose inputTokens breach the model's declared capacity.
|
|
91
|
+
const inferred = runtime.infer({ tokens });
|
|
92
|
+
if (!inferred.ok) {
|
|
93
|
+
return fallbackFromLoad({ ok: false, mode: "B", code: inferred.code }, reporter, tokens);
|
|
94
|
+
}
|
|
95
|
+
const vectorSet = encodeVectorSet(tokens, { reporter, seed: options.seed });
|
|
96
|
+
return { ok: true, mode: "A", vectorSet, code: null };
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Catch a (real or forced) A load failure (ok === false only) and select the
|
|
100
|
+
* B/C fallback that emits `vector_cortex_encoder_fallback_selected` from the
|
|
101
|
+
* production seam. The parameter is narrowed to the failed-load variant because
|
|
102
|
+
* the router hands off here only on a non-A/failed load — a qualified mode-A
|
|
103
|
+
* success never reaches this function (Q04).
|
|
104
|
+
*/
|
|
105
|
+
function fallbackFromLoad(loaded, reporter, tokens) {
|
|
106
|
+
if (loaded.mode === "C") {
|
|
107
|
+
const sel = selectLexicalC({ reporter });
|
|
108
|
+
const vector = embedLexical(textFromTokens(tokens));
|
|
109
|
+
return { ok: true, mode: "C", vector, width: sel.width, limitation: sel.limitation, code: loaded.code };
|
|
110
|
+
}
|
|
111
|
+
const sel = selectTrigramBFallback({ reporter });
|
|
112
|
+
const vector = embedTrigram512(textFromTokens(tokens));
|
|
113
|
+
return { ok: true, mode: "B", vector, width: sel.width, limitation: null, code: loaded.code };
|
|
114
|
+
}
|
|
115
|
+
export { ENC_FAIL };
|