lynkr 9.7.3 → 9.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +63 -25
  2. package/bin/cli.js +16 -1
  3. package/bin/lynkr-init.js +44 -1
  4. package/bin/lynkr-usage.js +78 -0
  5. package/bin/wrap.js +60 -35
  6. package/config/difficulty-anchors.json +22 -0
  7. package/package.json +23 -2
  8. package/scripts/audit-log-reader.js +399 -0
  9. package/scripts/build-eval-set.js +256 -0
  10. package/scripts/calibrate-thresholds.js +38 -157
  11. package/scripts/compact-dictionary.js +204 -0
  12. package/scripts/mine-difficulty-anchors.js +288 -0
  13. package/scripts/test-deduplication.js +448 -0
  14. package/scripts/validate-difficulty-classifier.js +123 -0
  15. package/scripts/validate-intent-anchors.js +186 -0
  16. package/scripts/ws7-anchor-replay.js +108 -0
  17. package/skills/lynkr/SKILL.md +195 -0
  18. package/src/api/middleware/loop-guard.js +87 -0
  19. package/src/api/middleware/request-logging.js +5 -64
  20. package/src/api/middleware/session.js +0 -0
  21. package/src/api/openai-router.js +120 -101
  22. package/src/api/providers-handler.js +27 -2
  23. package/src/api/router.js +467 -125
  24. package/src/budget/index.js +2 -19
  25. package/src/cache/semantic.js +9 -0
  26. package/src/clients/databricks.js +455 -142
  27. package/src/clients/gpt-utils.js +11 -105
  28. package/src/clients/openai-format.js +10 -3
  29. package/src/clients/openrouter-utils.js +49 -24
  30. package/src/clients/prompt-cache-injection.js +1 -0
  31. package/src/clients/provider-capabilities.js +1 -1
  32. package/src/clients/responses-format.js +34 -3
  33. package/src/clients/routing.js +15 -0
  34. package/src/config/index.js +36 -2
  35. package/src/context/gcf.js +275 -0
  36. package/src/context/tool-result-compressor.js +932 -47
  37. package/src/dashboard/api.js +1 -0
  38. package/src/logger/index.js +14 -1
  39. package/src/memory/search.js +12 -40
  40. package/src/memory/tools.js +3 -24
  41. package/src/orchestrator/bypass.js +4 -2
  42. package/src/orchestrator/index.js +120 -85
  43. package/src/routing/affinity-store.js +194 -0
  44. package/src/routing/agentic-detector.js +36 -6
  45. package/src/routing/bandit.js +25 -6
  46. package/src/routing/calibration.js +212 -0
  47. package/src/routing/classifier-setup.js +207 -0
  48. package/src/routing/client-profiles.js +292 -0
  49. package/src/routing/complexity-analyzer.js +88 -15
  50. package/src/routing/deescalator.js +148 -0
  51. package/src/routing/degradation.js +109 -0
  52. package/src/routing/difficulty-classifier.js +219 -0
  53. package/src/routing/feedback.js +157 -0
  54. package/src/routing/index.js +931 -90
  55. package/src/routing/intent-score.js +441 -0
  56. package/src/routing/interaction.js +3 -0
  57. package/src/routing/knn-router.js +70 -21
  58. package/src/routing/model-registry.js +28 -7
  59. package/src/routing/model-tiers.js +25 -2
  60. package/src/routing/reward-pipeline.js +68 -2
  61. package/src/routing/risk-analyzer.js +30 -1
  62. package/src/routing/risk-classifier.js +6 -2
  63. package/src/routing/session-affinity.js +162 -34
  64. package/src/routing/telemetry.js +286 -13
  65. package/src/routing/verifier.js +267 -0
  66. package/src/server.js +86 -21
  67. package/src/sessions/cleanup.js +17 -0
  68. package/src/tools/index.js +1 -15
  69. package/src/tools/smart-selection.js +10 -0
  70. package/src/tools/web-client.js +3 -3
  71. package/.eslintrc.cjs +0 -12
  72. package/benchmark-configs/litellm_config.yaml +0 -86
  73. package/benchmark-configs/lynkr.env +0 -48
  74. package/benchmark-configs/portkey-config.json +0 -60
  75. package/benchmark-configs/portkey-docker.sh +0 -23
  76. package/benchmark-tier-routing.js +0 -449
  77. package/funding.json +0 -110
  78. package/src/api/middleware/validation.js +0 -261
  79. package/src/routing/drift-monitor.js +0 -113
  80. package/src/workers/helpers.js +0 -185
@@ -32,6 +32,15 @@ let db = null;
32
32
  /** @type {boolean} */
33
33
  let initialised = false;
34
34
 
35
+ /**
36
+ * Test-only escape hatches. Production code should never touch these — the
37
+ * DB path is hardcoded to `<cwd>/.lynkr/telemetry.db`. Tests call these
38
+ * before the first `record()` to isolate their state.
39
+ * @type {string|null}
40
+ */
41
+ let _testDbPath = null;
42
+ let _testDbDisabled = false;
43
+
35
44
  /** Default retention: 30 days */
36
45
  const DEFAULT_RETENTION_MS = 30 * 24 * 60 * 60 * 1000;
37
46
 
@@ -49,16 +58,17 @@ function init() {
49
58
  }
50
59
 
51
60
  try {
52
- // Allow tests to redirect telemetry to an isolated DB so unit/integration
53
- // runs never pollute the production .lynkr/telemetry.db that build-knn-index.js
54
- // reads from. Empty string disables telemetry entirely.
55
- const override = process.env.LYNKR_TELEMETRY_DB_PATH;
61
+ // Path is hardcoded to <cwd>/.lynkr/telemetry.db in production. The
62
+ // pre-B `LYNKR_TELEMETRY_DB_PATH` env override was removed so operators
63
+ // can't accidentally divert telemetry to a stale path. Tests still need
64
+ // to isolate their DB — see `_setDbPathForTests` / `_disableForTests`
65
+ // below (module-scoped setters called before the first `record()`).
56
66
  let dbPath;
57
- if (override === "") {
58
- logger.debug("Telemetry: LYNKR_TELEMETRY_DB_PATH is empty, telemetry disabled");
67
+ if (_testDbDisabled) {
68
+ logger.debug("Telemetry: disabled for tests");
59
69
  return false;
60
- } else if (override) {
61
- dbPath = path.resolve(override);
70
+ } else if (_testDbPath) {
71
+ dbPath = path.resolve(_testDbPath);
62
72
  const overrideDir = path.dirname(dbPath);
63
73
  if (!fs.existsSync(overrideDir)) {
64
74
  fs.mkdirSync(overrideDir, { recursive: true });
@@ -111,7 +121,13 @@ function init() {
111
121
  tokens_per_second REAL,
112
122
  cost_efficiency REAL,
113
123
  request_text TEXT,
114
- response_text TEXT
124
+ response_text TEXT,
125
+ base_tier TEXT,
126
+ escalation_source TEXT,
127
+ propensity REAL,
128
+ candidates TEXT,
129
+ pinned INTEGER DEFAULT 0,
130
+ switch_reason TEXT
115
131
  );
116
132
 
117
133
  CREATE INDEX IF NOT EXISTS idx_telemetry_provider
@@ -125,14 +141,34 @@ function init() {
125
141
 
126
142
  CREATE INDEX IF NOT EXISTS idx_telemetry_session_id
127
143
  ON routing_telemetry(session_id, timestamp);
144
+
145
+ CREATE TABLE IF NOT EXISTS savings_events (
146
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
147
+ timestamp INTEGER NOT NULL,
148
+ category TEXT NOT NULL,
149
+ tokens_saved INTEGER NOT NULL
150
+ );
151
+
152
+ CREATE INDEX IF NOT EXISTS idx_savings_timestamp
153
+ ON savings_events(timestamp);
128
154
  `);
129
155
 
130
156
  // Migration: add columns to pre-existing tables (CREATE TABLE IF NOT EXISTS
131
157
  // won't add them to a DB created before these columns existed).
132
158
  const existingCols = new Set(db.prepare("PRAGMA table_info(routing_telemetry)").all().map((c) => c.name));
133
- for (const col of ["request_text", "response_text"]) {
159
+ const additiveCols = [
160
+ ["request_text", "TEXT"],
161
+ ["response_text", "TEXT"],
162
+ ["base_tier", "TEXT"],
163
+ ["escalation_source", "TEXT"],
164
+ ["propensity", "REAL"],
165
+ ["candidates", "TEXT"],
166
+ ["pinned", "INTEGER DEFAULT 0"],
167
+ ["switch_reason", "TEXT"],
168
+ ];
169
+ for (const [col, type] of additiveCols) {
134
170
  if (!existingCols.has(col)) {
135
- db.exec(`ALTER TABLE routing_telemetry ADD COLUMN ${col} TEXT`);
171
+ db.exec(`ALTER TABLE routing_telemetry ADD COLUMN ${col} ${type}`);
136
172
  }
137
173
  }
138
174
 
@@ -189,18 +225,27 @@ function record(data) {
189
225
  provider, model, routing_method, was_fallback, output_tokens,
190
226
  latency_ms, status_code, error_type, cost_usd, tool_calls_made,
191
227
  retry_count, circuit_breaker_state, quality_score, tokens_per_second,
192
- cost_efficiency, request_text, response_text
228
+ cost_efficiency, request_text, response_text,
229
+ base_tier, escalation_source, propensity, candidates, pinned, switch_reason
193
230
  ) VALUES (
194
231
  @request_id, @session_id, @timestamp, @complexity_score, @tier,
195
232
  @agentic_type, @tool_count, @input_tokens, @message_count, @request_type,
196
233
  @provider, @model, @routing_method, @was_fallback, @output_tokens,
197
234
  @latency_ms, @status_code, @error_type, @cost_usd, @tool_calls_made,
198
235
  @retry_count, @circuit_breaker_state, @quality_score, @tokens_per_second,
199
- @cost_efficiency, @request_text, @response_text
236
+ @cost_efficiency, @request_text, @response_text,
237
+ @base_tier, @escalation_source, @propensity, @candidates, @pinned, @switch_reason
200
238
  )`
201
239
  );
202
240
  if (!insert) return;
203
241
 
242
+ let candidatesJson = null;
243
+ if (data.candidates != null) {
244
+ candidatesJson = typeof data.candidates === "string"
245
+ ? data.candidates
246
+ : JSON.stringify(data.candidates);
247
+ }
248
+
204
249
  insert.run({
205
250
  request_id: data.request_id ?? null,
206
251
  session_id: data.session_id ?? null,
@@ -229,6 +274,12 @@ function record(data) {
229
274
  cost_efficiency: data.cost_efficiency ?? null,
230
275
  request_text: data.request_text ?? null,
231
276
  response_text: data.response_text ?? null,
277
+ base_tier: data.base_tier ?? null,
278
+ escalation_source: data.escalation_source ?? null,
279
+ propensity: data.propensity ?? null,
280
+ candidates: candidatesJson,
281
+ pinned: data.pinned ? 1 : 0,
282
+ switch_reason: data.switch_reason ?? null,
232
283
  });
233
284
  } catch (err) {
234
285
  logger.debug({ err: err.message }, "Telemetry record failed");
@@ -466,6 +517,121 @@ function getRoutingAccuracy(timeRange = {}) {
466
517
  }
467
518
  }
468
519
 
520
+ /**
521
+ * Aggregate escalation statistics.
522
+ *
523
+ * For every row where `escalation_source` is not null, returns a per-source
524
+ * breakdown: request count, total cost_usd, and total input_tokens (so
525
+ * callers can approximate a base-tier-vs-served-tier cost delta by
526
+ * combining with cost-optimizer.estimateCost for the two tiers).
527
+ *
528
+ * @param {Object} [timeRange]
529
+ * @param {number} [timeRange.since]
530
+ * @param {number} [timeRange.until]
531
+ * @returns {Object|null} { totalRequests, totalEscalated, escalatedPct, bySource: {source: {count, costUsd, inputTokens, avgQuality}} }
532
+ */
533
+ function getEscalationStats(timeRange = {}) {
534
+ if (!init()) return null;
535
+
536
+ const since = timeRange.since ?? Date.now() - 24 * 60 * 60 * 1000;
537
+ const until = timeRange.until ?? Date.now();
538
+
539
+ try {
540
+ const total = db
541
+ .prepare("SELECT COUNT(*) as cnt FROM routing_telemetry WHERE timestamp BETWEEN ? AND ?")
542
+ .get(since, until);
543
+ if (!total || total.cnt === 0) return null;
544
+
545
+ const rows = db
546
+ .prepare(
547
+ `SELECT
548
+ escalation_source as source,
549
+ COUNT(*) as cnt,
550
+ SUM(COALESCE(cost_usd, 0)) as cost_usd,
551
+ SUM(COALESCE(input_tokens, 0)) as input_tokens,
552
+ AVG(quality_score) as avg_quality
553
+ FROM routing_telemetry
554
+ WHERE timestamp BETWEEN ? AND ?
555
+ AND escalation_source IS NOT NULL
556
+ GROUP BY escalation_source`
557
+ )
558
+ .all(since, until);
559
+
560
+ const bySource = {};
561
+ let totalEscalated = 0;
562
+ for (const r of rows) {
563
+ bySource[r.source] = {
564
+ count: r.cnt,
565
+ costUsd: Math.round((r.cost_usd || 0) * 10000) / 10000,
566
+ inputTokens: r.input_tokens || 0,
567
+ avgQuality: r.avg_quality != null ? Math.round(r.avg_quality * 10) / 10 : null,
568
+ };
569
+ totalEscalated += r.cnt;
570
+ }
571
+
572
+ return {
573
+ totalRequests: total.cnt,
574
+ totalEscalated,
575
+ escalatedPct: Math.round((totalEscalated / total.cnt) * 1000) / 10,
576
+ bySource,
577
+ };
578
+ } catch (err) {
579
+ logger.debug({ err: err.message }, "Telemetry getEscalationStats failed");
580
+ return null;
581
+ }
582
+ }
583
+
584
+ /**
585
+ * Aggregate quality-by-tier-and-request-type — feeds the de-escalator's
586
+ * evidence check. Returns rows of {tier, request_type, count, avg_quality,
587
+ * error_rate}.
588
+ *
589
+ * @param {Object} [opts]
590
+ * @param {number} [opts.since] - ms epoch, defaults to 7 days ago
591
+ * @param {number} [opts.until] - defaults to now
592
+ * @param {string[]} [opts.tiers] - filter to specific tiers
593
+ * @returns {Array<{tier:string, request_type:string, count:number, avg_quality:number|null, error_rate:number}>|null}
594
+ */
595
+ function getQualityByTierAndType(opts = {}) {
596
+ if (!init()) return null;
597
+
598
+ const since = opts.since ?? Date.now() - 7 * 24 * 60 * 60 * 1000;
599
+ const until = opts.until ?? Date.now();
600
+
601
+ try {
602
+ const tierFilter = Array.isArray(opts.tiers) && opts.tiers.length > 0
603
+ ? `AND tier IN (${opts.tiers.map(() => "?").join(",")})`
604
+ : "";
605
+ const params = [since, until, ...(Array.isArray(opts.tiers) ? opts.tiers : [])];
606
+ const rows = db
607
+ .prepare(
608
+ `SELECT
609
+ tier,
610
+ request_type,
611
+ COUNT(*) as count,
612
+ AVG(quality_score) as avg_quality,
613
+ SUM(CASE WHEN error_type IS NOT NULL THEN 1 ELSE 0 END) * 1.0 / COUNT(*) as error_rate
614
+ FROM routing_telemetry
615
+ WHERE timestamp BETWEEN ? AND ?
616
+ AND tier IS NOT NULL
617
+ AND request_type IS NOT NULL
618
+ ${tierFilter}
619
+ GROUP BY tier, request_type`
620
+ )
621
+ .all(...params);
622
+ return rows.map((r) => ({
623
+ tier: r.tier,
624
+ request_type: r.request_type,
625
+ count: r.count,
626
+ avg_quality: r.avg_quality != null ? Math.round(r.avg_quality * 10) / 10 : null,
627
+ error_rate: Math.round(r.error_rate * 10000) / 10000,
628
+ }));
629
+ } catch (err) {
630
+ logger.debug({ err: err.message }, "Telemetry getQualityByTierAndType failed");
631
+ return null;
632
+ }
633
+ }
634
+
469
635
  /**
470
636
  * Delete telemetry records older than a given threshold.
471
637
  *
@@ -527,11 +693,118 @@ function getProviderStatsCached(provider, timeRange = {}) {
527
693
  return result;
528
694
  }
529
695
 
696
+ /**
697
+ * Return the shared telemetry sqlite handle (initialising if needed) so other
698
+ * routing subsystems can persist state alongside routing telemetry without
699
+ * opening a second WAL connection to the same file. Returns null when
700
+ * better-sqlite3 is unavailable or initialisation failed.
701
+ * @returns {import('better-sqlite3').Database|null}
702
+ */
703
+ function getDb() {
704
+ if (!init()) return null;
705
+ return db;
706
+ }
707
+
708
+ /**
709
+ * Record a token-savings event (tool stripping, compression, cache hit).
710
+ * Fire-and-forget like record(): never blocks or throws on the request path.
711
+ *
712
+ * @param {"tool_stripping"|"compression"|"cache_hit"} category
713
+ * @param {number} tokensSaved - Estimated tokens avoided (must be > 0 to record)
714
+ */
715
+ function recordSavings(category, tokensSaved) {
716
+ if (!Number.isFinite(tokensSaved) || tokensSaved <= 0) return;
717
+ if (!init()) return;
718
+
719
+ setImmediate(() => {
720
+ try {
721
+ const insert = stmt(
722
+ "insertSavings",
723
+ `INSERT INTO savings_events (timestamp, category, tokens_saved)
724
+ VALUES (@timestamp, @category, @tokens_saved)`
725
+ );
726
+ if (!insert) return;
727
+ insert.run({
728
+ timestamp: Date.now(),
729
+ category: String(category),
730
+ tokens_saved: Math.round(tokensSaved),
731
+ });
732
+ } catch (err) {
733
+ logger.debug({ err: err.message }, "Failed to record savings event");
734
+ }
735
+ });
736
+ }
737
+
738
+ /**
739
+ * Summarise savings events since a timestamp.
740
+ *
741
+ * @param {number} sinceMs - Epoch ms lower bound (0 for all time)
742
+ * @returns {{total: number, byCategory: Object<string, number>}}
743
+ */
744
+ function getSavingsSummary(sinceMs = 0) {
745
+ const empty = { total: 0, byCategory: {} };
746
+ if (!init()) return empty;
747
+ try {
748
+ const rows = db
749
+ .prepare(
750
+ `SELECT category, SUM(tokens_saved) AS tokens
751
+ FROM savings_events WHERE timestamp >= ? GROUP BY category`
752
+ )
753
+ .all(sinceMs);
754
+ const byCategory = {};
755
+ let total = 0;
756
+ for (const row of rows) {
757
+ byCategory[row.category] = row.tokens || 0;
758
+ total += row.tokens || 0;
759
+ }
760
+ return { total, byCategory };
761
+ } catch (err) {
762
+ logger.debug({ err: err.message }, "Failed to read savings summary");
763
+ return empty;
764
+ }
765
+ }
766
+
767
+ /**
768
+ * Test-only helpers. Do NOT call from production code. These exist because
769
+ * the DB path was hardcoded (no more `LYNKR_TELEMETRY_DB_PATH` env var), so
770
+ * tests need an alternative way to route telemetry at an isolated file.
771
+ *
772
+ * Both helpers must be called BEFORE the first `record()` call — after
773
+ * that the DB handle is memoised and won't re-open. Call `_resetForTests`
774
+ * between tests to force re-initialisation.
775
+ */
776
+ function _setDbPathForTests(p) {
777
+ _testDbPath = p || null;
778
+ _testDbDisabled = false;
779
+ }
780
+ function _disableForTests() {
781
+ _testDbDisabled = true;
782
+ _testDbPath = null;
783
+ }
784
+ function _resetForTests() {
785
+ if (db) {
786
+ try { db.close(); } catch { /* ignore */ }
787
+ }
788
+ db = null;
789
+ initialised = false;
790
+ _testDbPath = null;
791
+ _testDbDisabled = false;
792
+ }
793
+
530
794
  module.exports = {
531
795
  record,
532
796
  query,
533
797
  getStats: getStatsCached,
534
798
  getProviderStats: getProviderStatsCached,
535
799
  getRoutingAccuracy,
800
+ getEscalationStats,
801
+ getQualityByTierAndType,
802
+ recordSavings,
803
+ getSavingsSummary,
536
804
  cleanup,
805
+ getDb,
806
+ // Test-only — do not use in production code.
807
+ _setDbPathForTests,
808
+ _disableForTests,
809
+ _resetForTests,
537
810
  };
@@ -0,0 +1,267 @@
1
+ /**
2
+ * WS6 — response verifier for cascade routing.
3
+ *
4
+ * Judges whether a CHEAP-tier response is good enough to serve, so the
5
+ * cascade can discard bad answers and escalate instead of delivering them.
6
+ * Rationale (deep-research, 2026-07-08): ex-ante difficulty prediction
7
+ * failed to beat trivial baselines on agentic coding traffic (SWE-Bench),
8
+ * while try-cheap-then-verify cascades gained up to 14%. The verifier is
9
+ * where that gain lives.
10
+ *
11
+ * Design constraints:
12
+ * - Layer 1: deterministic structural checks targeting the cheap-model
13
+ * failure modes observed live — language drift (CJK mid-English),
14
+ * degeneration loops, truncation, malformed tool calls, empty/echo
15
+ * output. High precision: flag only what is definitely broken.
16
+ * - Layer 2: a coarse content-quality score with a conservative
17
+ * threshold. Catches low-effort responses to substantive asks.
18
+ * - No LLM-judge in v1 (research: unvalidated self-assessment is weak),
19
+ * no logprobs (Ollama's Anthropic passthrough doesn't expose them).
20
+ * A confidently-wrong but fluent answer WILL pass — the target is
21
+ * garbage, not falsehood.
22
+ * - Pure function, no I/O. Never throws: any internal error returns
23
+ * verdict "pass" (fail-open — a broken verifier must not break
24
+ * serving or force spurious escalations).
25
+ *
26
+ * Verification only ever applies to cheap-tier responses; expensive-tier
27
+ * answers are never second-guessed (caller enforces).
28
+ */
29
+
30
+ const logger = require('../logger');
31
+
32
+ // ---------------------------------------------------------------------------
33
+ // Text extraction
34
+ // ---------------------------------------------------------------------------
35
+
36
+ function _responseText(responseBody) {
37
+ const content = responseBody?.content;
38
+ if (typeof content === 'string') return content;
39
+ if (!Array.isArray(content)) return '';
40
+ return content
41
+ .filter((b) => b?.type === 'text' && typeof b.text === 'string')
42
+ .map((b) => b.text)
43
+ .join('\n');
44
+ }
45
+
46
+ function _lastUserText(payload) {
47
+ const msgs = payload?.messages;
48
+ if (!Array.isArray(msgs)) return '';
49
+ for (let i = msgs.length - 1; i >= 0; i--) {
50
+ const m = msgs[i];
51
+ if (m?.role !== 'user') continue;
52
+ const raw = typeof m.content === 'string'
53
+ ? m.content
54
+ : Array.isArray(m.content)
55
+ ? m.content.filter((b) => b?.type === 'text').map((b) => b.text || '').join(' ')
56
+ : '';
57
+ return raw
58
+ .replace(/<system-reminder>[\s\S]*?<\/system-reminder>/g, ' ')
59
+ // Codex harness blocks merged into the typed text — inflate askLen
60
+ // and trip the wants-code regexes ("claude-code", access="write"),
61
+ // failing every short answer to a trivial prompt.
62
+ .replace(/<environment_context>[\s\S]*?<\/environment_context>/g, ' ')
63
+ .replace(/<user_instructions>[\s\S]*?<\/user_instructions>/g, ' ')
64
+ // Goose wraps every typed message in a turn-context block (time, cwd,
65
+ // todo notes) — 275+ chars of harness plumbing that made a bare "Hi"
66
+ // look like a substantive ask, failing every short cheap-tier answer.
67
+ .replace(/<turn-context>[\s\S]*?<\/turn-context>/g, ' ')
68
+ .trim();
69
+ }
70
+ return '';
71
+ }
72
+
73
+ // ---------------------------------------------------------------------------
74
+ // Layer 1 — structural checks (each returns a reason string or null)
75
+ // ---------------------------------------------------------------------------
76
+
77
+ /**
78
+ * Language drift: unexpected CJK/Cyrillic content in a conversation whose
79
+ * user text contains none. Live incident: minimax emitting "+统一接口:" in
80
+ * the middle of an English refactor plan. Threshold is a RATIO so quoting
81
+ * a foreign identifier from the repo doesn't trip it.
82
+ */
83
+ function checkLanguageDrift(userText, answerText) {
84
+ if (!answerText) return null;
85
+ const FOREIGN = /[一-鿿぀-ヿ가-힯Ѐ-ӿ]/g;
86
+ const userForeign = (userText.match(FOREIGN) || []).length;
87
+ if (userForeign > 0) return null; // user writes that script — anything goes
88
+ const answerForeign = (answerText.match(FOREIGN) || []).length;
89
+ if (answerForeign === 0) return null;
90
+ const ratio = answerForeign / Math.max(1, answerText.length);
91
+ // A couple of quoted characters is fine; sustained drift is not.
92
+ if (answerForeign >= 6 || ratio > 0.02) {
93
+ return `language-drift (${answerForeign} foreign chars, user text had none)`;
94
+ }
95
+ return null;
96
+ }
97
+
98
+ /**
99
+ * Degeneration: the same shingle repeating far beyond what natural text
100
+ * (or even a list) produces. Catches small-model repetition loops.
101
+ */
102
+ function checkDegeneration(answerText) {
103
+ if (!answerText || answerText.length < 200) return null;
104
+ const words = answerText.toLowerCase().split(/\s+/).filter(Boolean);
105
+ if (words.length < 40) return null;
106
+ const SHINGLE = 5;
107
+ const counts = new Map();
108
+ for (let i = 0; i + SHINGLE <= words.length; i++) {
109
+ const key = words.slice(i, i + SHINGLE).join(' ');
110
+ counts.set(key, (counts.get(key) || 0) + 1);
111
+ }
112
+ let max = 0;
113
+ for (const c of counts.values()) if (c > max) max = c;
114
+ const total = Math.max(1, words.length - SHINGLE + 1);
115
+ if (max >= 5 && max / total > 0.08) {
116
+ return `degeneration (a 5-gram repeats ${max}× across ${total} positions)`;
117
+ }
118
+ return null;
119
+ }
120
+
121
+ /**
122
+ * Truncation: response ran into max_tokens mid-structure. A length stop
123
+ * by itself is common (long answers); a length stop with an UNCLOSED code
124
+ * fence means the useful part is cut.
125
+ */
126
+ function checkTruncation(answerText, responseBody) {
127
+ if (responseBody?.stop_reason !== 'max_tokens' && responseBody?.stop_reason !== 'length') return null;
128
+ const fences = (answerText.match(/```/g) || []).length;
129
+ if (fences % 2 === 1) {
130
+ return 'truncation (hit token limit inside an open code fence)';
131
+ }
132
+ return null;
133
+ }
134
+
135
+ /**
136
+ * Malformed tool calls: tool_use blocks whose input is not an object or
137
+ * whose name is missing. Live symptom: Claude Code's "Invalid tool
138
+ * parameters" errors on cheap-model responses.
139
+ */
140
+ function checkMalformedToolCalls(responseBody) {
141
+ const content = responseBody?.content;
142
+ if (!Array.isArray(content)) return null;
143
+ for (const b of content) {
144
+ if (b?.type !== 'tool_use') continue;
145
+ if (!b.name || typeof b.name !== 'string') return 'malformed-tool-call (missing name)';
146
+ if (b.input === undefined || b.input === null || typeof b.input !== 'object' || Array.isArray(b.input)) {
147
+ return `malformed-tool-call (input is ${Array.isArray(b.input) ? 'array' : typeof b.input})`;
148
+ }
149
+ }
150
+ return null;
151
+ }
152
+
153
+ /** Empty or echo output. */
154
+ function checkEmptyOrEcho(userText, answerText, responseBody) {
155
+ const hasToolUse = Array.isArray(responseBody?.content)
156
+ && responseBody.content.some((b) => b?.type === 'tool_use');
157
+ if (hasToolUse) return null; // tool-only turns legitimately carry no prose
158
+ const t = (answerText || '').trim();
159
+ if (t.length < 2) return 'empty-response';
160
+ if (userText.length > 40 && t.length > 40) {
161
+ const a = t.toLowerCase().slice(0, 200);
162
+ const u = userText.toLowerCase().slice(0, 200);
163
+ if (a === u) return 'prompt-echo';
164
+ }
165
+ return null;
166
+ }
167
+
168
+ // ---------------------------------------------------------------------------
169
+ // Layer 2 — coarse content-quality score
170
+ // ---------------------------------------------------------------------------
171
+
172
+ /**
173
+ * 0-100. Conservative: only very low scores fail (threshold below).
174
+ * Signals: effort proportional to the ask, structure when structure was
175
+ * requested, code when code was requested.
176
+ */
177
+ function contentScore(userText, answerText, responseBody) {
178
+ const hasToolUse = Array.isArray(responseBody?.content)
179
+ && responseBody.content.some((b) => b?.type === 'tool_use');
180
+ if (hasToolUse) return 100; // agent turns are judged by their tool calls, not prose
181
+
182
+ let score = 60; // neutral prior
183
+ const askLen = userText.length;
184
+ const ansLen = (answerText || '').length;
185
+
186
+ // Effort: a substantive ask answered in a stub.
187
+ if (askLen > 200 && ansLen < 60) score -= 35;
188
+ else if (askLen > 100 && ansLen < 30) score -= 30;
189
+ else if (ansLen >= 120) score += 15;
190
+
191
+ // Structure requested → structure delivered?
192
+ const wantsStructure = /\b(list|steps?|plan|compare|table|pros and cons|trade-?offs)\b/i.test(userText);
193
+ const hasStructure = /(^|\n)\s*([-*•]|\d+[.)])\s+|\n#{1,3}\s|\|.*\|/.test(answerText || '');
194
+ if (wantsStructure) score += hasStructure ? 15 : -20;
195
+
196
+ // Code requested → code delivered?
197
+ const wantsCode = /\b(code|function|implement|snippet|example|replacement|fix)\b/i.test(userText)
198
+ && /```/.test(userText || '') === false; // asking about pasted code still often warrants code back — keep loose
199
+ const hasCode = /```|(^|\n) {4}\S/.test(answerText || '');
200
+ if (wantsCode && /\b(write|implement|show|give me)\b/i.test(userText)) {
201
+ score += hasCode ? 10 : -15;
202
+ }
203
+
204
+ return Math.max(0, Math.min(100, score));
205
+ }
206
+
207
+ // ---------------------------------------------------------------------------
208
+ // Public API
209
+ // ---------------------------------------------------------------------------
210
+
211
+ const CONTENT_SCORE_FAIL_THRESHOLD = 30;
212
+
213
+ /**
214
+ * @param {object} args
215
+ * @param {object} args.payload — the request payload
216
+ * @param {object} args.responseBody — Anthropic-format response body
217
+ * @returns {{verdict: 'pass'|'fail', score: number|null, reasons: string[]}}
218
+ */
219
+ function verify({ payload, responseBody } = {}) {
220
+ try {
221
+ // No body / unrecognizable content shape = nothing to verify (upstream
222
+ // error paths have their own handling). Verification judges answers,
223
+ // not absences or shapes we don't understand.
224
+ if (!responseBody || typeof responseBody !== 'object') {
225
+ return { verdict: 'pass', score: null, reasons: [] };
226
+ }
227
+ const c = responseBody.content;
228
+ if (typeof c !== 'string' && !Array.isArray(c)) {
229
+ return { verdict: 'pass', score: null, reasons: [] };
230
+ }
231
+ const userText = _lastUserText(payload);
232
+ const answerText = _responseText(responseBody);
233
+
234
+ const reasons = [
235
+ checkLanguageDrift(userText, answerText),
236
+ checkDegeneration(answerText),
237
+ checkTruncation(answerText, responseBody),
238
+ checkMalformedToolCalls(responseBody),
239
+ checkEmptyOrEcho(userText, answerText, responseBody),
240
+ ].filter(Boolean);
241
+
242
+ const score = contentScore(userText, answerText, responseBody);
243
+ if (reasons.length === 0 && score < CONTENT_SCORE_FAIL_THRESHOLD) {
244
+ reasons.push(`low-content-score (${score} < ${CONTENT_SCORE_FAIL_THRESHOLD})`);
245
+ }
246
+
247
+ return { verdict: reasons.length ? 'fail' : 'pass', score, reasons };
248
+ } catch (err) {
249
+ // Fail-open: a broken verifier must never block serving.
250
+ logger.debug({ err: err.message }, '[Verifier] error — failing open');
251
+ return { verdict: 'pass', score: null, reasons: [] };
252
+ }
253
+ }
254
+
255
+ module.exports = {
256
+ verify,
257
+ CONTENT_SCORE_FAIL_THRESHOLD,
258
+ // exported for unit tests
259
+ _internal: {
260
+ checkLanguageDrift,
261
+ checkDegeneration,
262
+ checkTruncation,
263
+ checkMalformedToolCalls,
264
+ checkEmptyOrEcho,
265
+ contentScore,
266
+ },
267
+ };