thumbgate 1.34.3 → 1.37.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/cyberstrike-compare-not-clone/SKILL.md +36 -0
- package/.agents/skills/gitlab-sandbox-allowlist-not-trust/SKILL.md +77 -0
- package/.agents/skills/jit-harness-compare-not-clone/SKILL.md +34 -0
- package/.agents/skills/openui-catalog-compose-honesty/SKILL.md +64 -0
- package/.agents/skills/zvec-grep-compare-not-clone/SKILL.md +34 -0
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/llms.txt +1 -0
- package/.well-known/mcp/server-card.json +1 -1
- package/CONTRIBUTING.md +95 -0
- package/README.md +195 -632
- package/THIRD_PARTY_NOTICES.md +89 -0
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/future-agi/.mcp.json +8 -0
- package/adapters/future-agi/FUTURE_AGI.md +23 -0
- package/adapters/future-agi/config.toml +3 -0
- package/adapters/future-agi/future-agi-bridge.js +9 -0
- package/adapters/future-agi/opencode.json +8 -0
- package/adapters/herdr/herdr-plugin.toml +18 -0
- package/adapters/mcp/server-stdio.js +238 -25
- package/adapters/opencode/opencode.json +1 -1
- package/adapters/workos/WORKOS.md +52 -0
- package/bin/cli.js +373 -5
- package/bin/futureagi-bridge +9 -0
- package/config/gate-templates.json +653 -4
- package/config/gates/actor-critic-audit.json +34 -0
- package/config/gates/default.json +21 -2
- package/config/gates/five-walls-governance.json +34 -0
- package/config/gates/future-agi-guardrails.json +34 -0
- package/config/gates/radware-threat-defense-2026.json +61 -0
- package/config/gates/simatree-data-governance.json +33 -0
- package/config/mcp-allowlists.json +4 -0
- package/config/merge-quality-checks.json +10 -1
- package/config/model-candidates.json +382 -24
- package/config/model-tiers.json +18 -0
- package/config/post-deploy-marketing-pages.json +10 -0
- package/config/progressive/01-wire-only.json +11 -0
- package/config/progressive/02-dashboard-empty-ok.json +10 -0
- package/config/progressive/03-one-lesson.json +10 -0
- package/config/progressive/04-warn-fires.json +11 -0
- package/config/progressive/05-strict-optional.json +11 -0
- package/config/progressive/README.md +15 -0
- package/config/schemas/broker-execution-receipt.schema.json +139 -0
- package/config/schemas/provider-execution-attestation-v1.schema.json +58 -0
- package/conformance/provider-attestation/vectors.json +320 -0
- package/docs/specs/provider-execution-attestation-v1.md +69 -0
- package/openapi/openapi.yaml +15 -0
- package/package.json +401 -147
- package/public/about.html +2 -2
- package/public/ai-malpractice-prevention.html +7 -7
- package/public/blog/a-10-dollar-vps-is-not-a-computer.html +143 -0
- package/public/blog/a-receipt-is-not-world-state.html +388 -0
- package/public/blog/git-at-agent-scale.html +374 -0
- package/public/blog/no-llm-in-the-gate.html +133 -0
- package/public/blog.html +80 -0
- package/public/case-studies.html +16 -1
- package/public/compare.html +28 -0
- package/public/diagnostic.html +216 -7
- package/public/docs/connectors.html +39 -0
- package/public/federal.html +2 -2
- package/public/founders.html +639 -0
- package/public/index.html +87 -9
- package/public/install.html +8 -8
- package/public/learn.html +39 -0
- package/public/numbers.html +2 -2
- package/public/peter.html +310 -0
- package/public/platform-partners.html +119 -0
- package/public/pricing.html +24 -3
- package/public/privacy.html +117 -0
- package/public/pro.html +17 -0
- package/public/support.html +62 -0
- package/public/terms.html +130 -0
- package/public/third-party-notices.html +95 -0
- package/public/yt.html +351 -0
- package/scripts/action-receipts.js +133 -3
- package/scripts/adaptive-governance-arena.js +349 -0
- package/scripts/admin-override.js +205 -0
- package/scripts/agent-action-inventory.js +869 -0
- package/scripts/agent-audit-trace.js +42 -2
- package/scripts/agent-egress-policy.js +1117 -0
- package/scripts/agent-memory-lifecycle.js +141 -2
- package/scripts/agent-operations-planner.js +441 -1
- package/scripts/agent-readiness.js +68 -0
- package/scripts/agent-security-central.js +647 -0
- package/scripts/allowlist-bridge-honesty.js +417 -0
- package/scripts/async-job-runner.js +102 -11
- package/scripts/audit-trail.js +212 -0
- package/scripts/auto-promote-gates.js +178 -27
- package/scripts/billing.js +1 -1
- package/scripts/broker-execution-receipts.js +719 -0
- package/scripts/budget-aware-gates-proof.js +423 -0
- package/scripts/claude-feedback-sync.js +29 -3
- package/scripts/claw-harness-production.js +237 -0
- package/scripts/cli-progress.js +111 -0
- package/scripts/cli-schema.js +163 -1
- package/scripts/codex-runbook-flywheel.js +318 -0
- package/scripts/context-footprint.js +186 -0
- package/scripts/contextfs.js +143 -61
- package/scripts/dashboard-limits.js +27 -0
- package/scripts/dashboard.js +279 -9
- package/scripts/deepseek-v4-runtime-guardrails.js +72 -6
- package/scripts/docker-sandbox-planner.js +18 -0
- package/scripts/double-blind-eval-protocol.js +252 -0
- package/scripts/edotenv-rl-gateway.js +259 -0
- package/scripts/ensure-production-search-corpus.js +162 -0
- package/scripts/eval-holdout.js +311 -0
- package/scripts/feedback-aggregate.js +21 -2
- package/scripts/feedback-loop.js +87 -5
- package/scripts/feedback-quality.js +9 -0
- package/scripts/file-ledger-lock.js +4 -1
- package/scripts/financial-control-plane.js +41 -1
- package/scripts/find-dormant-requires.js +118 -0
- package/scripts/fs-utils.js +84 -8
- package/scripts/gate-stats.js +2 -2
- package/scripts/gates-engine.js +859 -58
- package/scripts/generate-case-study-outreach.js +24 -15
- package/scripts/git-at-scale.js +628 -0
- package/scripts/governance-conflict-audit.js +1650 -0
- package/scripts/governance-difficulty-curriculum.js +328 -0
- package/scripts/graphrag-retrieval.js +275 -0
- package/scripts/gurobi-optimizer.js +324 -0
- package/scripts/gurobi_optimizer.py +485 -0
- package/scripts/harness-selector.js +82 -1
- package/scripts/hidden-entry-points.js +284 -0
- package/scripts/human-escalation.js +199 -1
- package/scripts/hybrid-feedback-context.js +152 -19
- package/scripts/intent-governed-execution.js +602 -0
- package/scripts/intervention-policy.js +123 -20
- package/scripts/jit-harness-compose.js +628 -0
- package/scripts/jsonl-watcher.js +10 -0
- package/scripts/lesson-embedding-index.js +95 -12
- package/scripts/lesson-retrieval.js +105 -19
- package/scripts/local-model-profile.js +19 -2
- package/scripts/mailer/resend-mailer.js +1 -1
- package/scripts/matryoshka-embedding.js +235 -0
- package/scripts/mcp-oauth.js +42 -4
- package/scripts/mcp-session-handles.js +1016 -0
- package/scripts/mcp-wiring-doctor.js +314 -0
- package/scripts/memory-firewall.js +115 -2
- package/scripts/memory-scope-readiness.js +299 -0
- package/scripts/memory-vs-rag-route.js +161 -0
- package/scripts/model-tier-router.js +148 -21
- package/scripts/nvidia-specdecode-al-doctor.js +536 -0
- package/scripts/openui-catalog-compose-honesty.js +593 -0
- package/scripts/operational-integrity.js +19 -1
- package/scripts/override-audit.js +213 -0
- package/scripts/package-manager-honesty-doctor.js +458 -0
- package/scripts/pr-manager.js +63 -1
- package/scripts/prove-herdr-adapter.js +52 -0
- package/scripts/prove-memory-pyramid-and-symbolic-canvas.js +95 -0
- package/scripts/prove-workos.js +73 -0
- package/scripts/provider-attestation-conformance.js +192 -0
- package/scripts/provider-receipt-contract.js +136 -0
- package/scripts/qwen38-max-cost-optimizer.js +401 -0
- package/scripts/radware-threat-defense.js +280 -0
- package/scripts/rag-embedding-identity.js +221 -0
- package/scripts/rag-precision-guardrails.js +112 -2
- package/scripts/remote-feedback-capture.js +159 -0
- package/scripts/research-agent-harness.js +256 -0
- package/scripts/rsi-safety-hillclimb.js +200 -0
- package/scripts/rule-sprawl.js +188 -0
- package/scripts/schedule-manager.js +147 -0
- package/scripts/self-heal.js +8 -0
- package/scripts/session-lease.js +415 -0
- package/scripts/simatree-data-governance.js +347 -0
- package/scripts/slo-alert-engine.js +172 -7
- package/scripts/solver-parity.js +539 -0
- package/scripts/stealth-memory-injection-gate.js +333 -0
- package/scripts/switchyard-router.js +366 -0
- package/scripts/telemetry-analytics.js +84 -27
- package/scripts/temporal-decay-weighting.js +138 -0
- package/scripts/test-all.js +165 -0
- package/scripts/token-savings.js +42 -0
- package/scripts/tool-kpi-tracker.js +108 -5
- package/scripts/tool-registry.js +193 -5
- package/scripts/universal-claim-evaluator.js +14 -2
- package/scripts/vector-store.js +279 -9
- package/scripts/workflow-notebook.js +391 -0
- package/scripts/workflow-sentinel.js +111 -12
- package/scripts/workos-production-guard.js +260 -0
- package/scripts/workspace-search-route.js +515 -0
- package/server.json +2 -2
- package/src/agent-identity-boundary.js +76 -0
- package/src/agent-retrieval-cache.js +155 -0
- package/src/alert-noise-ledger.js +502 -0
- package/src/api/server.js +802 -185
- package/src/git-fast-cache.js +220 -0
- package/src/git-wal-sync.js +156 -0
- package/src/hash-anchored-edit.js +82 -0
- package/src/hermes-platform-protocol.js +475 -0
- package/src/hermes-sync-plane.js +241 -0
- package/src/index.js +30 -1
- package/src/iso42001-compliance-guard.js +97 -0
- package/src/latency-budget.js +244 -0
- package/src/mcp-writeguard.js +316 -0
- package/src/miminions-adapter.js +106 -0
- package/src/pipeline-compass.js +104 -0
- package/src/ppl-alert-pipeline.js +284 -0
- package/src/rendezvous-router.js +90 -0
- package/src/security-questionnaire.js +195 -0
|
@@ -176,13 +176,51 @@ function patternContext(entry) {
|
|
|
176
176
|
return context;
|
|
177
177
|
}
|
|
178
178
|
|
|
179
|
+
/**
|
|
180
|
+
* Tags that mark capture/enforcement machinery, not a human judgment about an
|
|
181
|
+
* agent action. An entry carrying one of these is never a human lesson.
|
|
182
|
+
*/
|
|
183
|
+
const MACHINERY_FEEDBACK_TAGS = Object.freeze([
|
|
184
|
+
'auto-capture',
|
|
185
|
+
'gates-engine',
|
|
186
|
+
'audit-trail',
|
|
187
|
+
]);
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Transport tags stamped by `scripts/claude-feedback-sync.js` on EVERY record
|
|
191
|
+
* it recovers from Claude history — bare "thumbs down" junk and rich human
|
|
192
|
+
* lessons alike. Bare rows were measured (2026-08-26) becoming the PreToolUse
|
|
193
|
+
* constraint `Avoid: "thumbs down claude-history-sync auto-capture-fallback"
|
|
194
|
+
* (seen 44x)`. Because the tags mark the transport rather than the content,
|
|
195
|
+
* they demote an entry to "automated" only when the recovered text is a bare
|
|
196
|
+
* signal with no described mistake; substantive repeated feedback such as
|
|
197
|
+
* "thumbs down: claimed merged without a SHA" must still rank.
|
|
198
|
+
*/
|
|
199
|
+
const HISTORY_SYNC_TRANSPORT_TAGS = Object.freeze([
|
|
200
|
+
'auto-capture-fallback',
|
|
201
|
+
'claude-history-sync',
|
|
202
|
+
]);
|
|
203
|
+
|
|
204
|
+
const BARE_SIGNAL_PREFIX_RE = /^(?:thumbs\s*(?:up|down)|👍|👎)[\s:,.!-]*/i;
|
|
205
|
+
|
|
206
|
+
function isBareHistorySyncSignal(entry) {
|
|
207
|
+
if (String(entry.whatToChange || entry.what_to_change || '').trim()) return false;
|
|
208
|
+
const text = String(entry.context || entry.whatWentWrong || entry.what_went_wrong || '').trim();
|
|
209
|
+
if (!text) return true;
|
|
210
|
+
const stripped = text.replace(BARE_SIGNAL_PREFIX_RE, '').trim();
|
|
211
|
+
return stripped.length < 8;
|
|
212
|
+
}
|
|
213
|
+
|
|
179
214
|
/**
|
|
180
215
|
* Check if the feedback entry is an automated enforcement log (e.g. from gates engine)
|
|
181
216
|
* rather than real developer/user feedback.
|
|
182
217
|
*/
|
|
183
218
|
function isAutomatedFeedback(entry) {
|
|
184
|
-
const tags = entry.tags
|
|
185
|
-
if (tags.
|
|
219
|
+
const tags = Array.isArray(entry.tags) ? entry.tags : [];
|
|
220
|
+
if (tags.some((tag) => MACHINERY_FEEDBACK_TAGS.includes(tag))) {
|
|
221
|
+
return true;
|
|
222
|
+
}
|
|
223
|
+
if (tags.some((tag) => HISTORY_SYNC_TRANSPORT_TAGS.includes(tag)) && isBareHistorySyncSignal(entry)) {
|
|
186
224
|
return true;
|
|
187
225
|
}
|
|
188
226
|
const context = String(entry.context || entry.whatWentWrong || '').toLowerCase();
|
|
@@ -203,15 +241,51 @@ function getTimestampMs(value) {
|
|
|
203
241
|
* Extract meaningful keywords from text.
|
|
204
242
|
* min 4 chars, no stopwords, max 8 tokens.
|
|
205
243
|
*/
|
|
244
|
+
// Hook payloads are JSON envelopes. When a malformed one leaks into the feedback
|
|
245
|
+
// store, its FIELD NAMES become "keywords" — and field names appear in every
|
|
246
|
+
// payload, so the resulting pattern matches every action ever taken. One corrupt
|
|
247
|
+
// entry becomes a universal block.
|
|
248
|
+
//
|
|
249
|
+
// Observed 2026-08-06: a truncated payload fragment
|
|
250
|
+
// { , , ,"workspaceroot":"/users/<redacted>/workspace/git/igor/thumbgate/", , , ,"pe"
|
|
251
|
+
// was admitted as a recurring negative pattern with count 6. "workspaceroot" is
|
|
252
|
+
// long enough that isSpecificKeyword() treats it as decisive on its own, so a
|
|
253
|
+
// SINGLE hit hard-denied every Bash call in the repository — including the
|
|
254
|
+
// commands needed to diagnose it. It blocked 20 times and warned zero times.
|
|
255
|
+
//
|
|
256
|
+
// These tokens describe the transport, never the mistake, so they can never be
|
|
257
|
+
// evidence of a recurring failure.
|
|
258
|
+
const ENVELOPE_TOKENS = new Set([
|
|
259
|
+
'workspaceroot', 'toolname', 'toolinput', 'tooloutput', 'toolresponse', 'tooluseid',
|
|
260
|
+
'sessionid', 'transcriptpath', 'hookeventname', 'permissionmode', 'promptid',
|
|
261
|
+
'cwd', 'timestamp', 'metadata', 'payload', 'stdout', 'stderr',
|
|
262
|
+
'users', 'workspace', 'thumbgate', 'igor',
|
|
263
|
+
]);
|
|
264
|
+
|
|
206
265
|
function keywords(text) {
|
|
207
266
|
if (!text) return [];
|
|
208
267
|
const tokens = normalize(text)
|
|
209
268
|
.replace(/[^a-z0-9\s_-]/g, ' ')
|
|
210
269
|
.split(/\s+/)
|
|
211
|
-
.filter((t) => t.length >= 4 && !STOPWORDS.has(t));
|
|
270
|
+
.filter((t) => t.length >= 4 && !STOPWORDS.has(t) && !ENVELOPE_TOKENS.has(t));
|
|
212
271
|
return [...new Set(tokens)].slice(0, 8);
|
|
213
272
|
}
|
|
214
273
|
|
|
274
|
+
/**
|
|
275
|
+
* True when text is a serialized payload fragment rather than a description of
|
|
276
|
+
* a mistake. Real lessons are prose; fragments are punctuation and field names.
|
|
277
|
+
* Admitting one as a pattern is how a transport artifact becomes a gate.
|
|
278
|
+
*/
|
|
279
|
+
function looksLikeSerializedFragment(text) {
|
|
280
|
+
const raw = String(text || '');
|
|
281
|
+
if (!raw.trim()) return false;
|
|
282
|
+
// Runs of empty JSON slots (", , ,") only occur when a serializer dropped keys.
|
|
283
|
+
if (/(?:,\s*){3,}/.test(raw)) return true;
|
|
284
|
+
const wordChars = (raw.match(/[a-z]/gi) || []).length;
|
|
285
|
+
const structural = (raw.match(/["{}[\]:,]/g) || []).length;
|
|
286
|
+
return wordChars > 0 && structural >= wordChars / 2;
|
|
287
|
+
}
|
|
288
|
+
|
|
215
289
|
/**
|
|
216
290
|
* FNV-1a 32-bit hash.
|
|
217
291
|
*/
|
|
@@ -277,11 +351,12 @@ function buildHybridState(opts) {
|
|
|
277
351
|
if (cls === 'positive') positive++;
|
|
278
352
|
if (cls === 'negative') {
|
|
279
353
|
negative++;
|
|
280
|
-
//
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
354
|
+
// History-sync fallback and gate logs still count as events, but they must
|
|
355
|
+
// not become recurring "Avoid" constraints injected on every PreToolUse.
|
|
356
|
+
if (isAutomatedFeedback(entry)) continue;
|
|
357
|
+
|
|
358
|
+
const toolName = inferToolName(entry.toolName || entry.tool_name || 'unknown', entry.context || '');
|
|
359
|
+
toolNegatives[toolName] = (toolNegatives[toolName] || 0) + 1;
|
|
285
360
|
|
|
286
361
|
// Build pattern from context / whatWentWrong / what_went_wrong
|
|
287
362
|
const rawText = [
|
|
@@ -346,6 +421,8 @@ function buildHybridState(opts) {
|
|
|
346
421
|
].join(' ');
|
|
347
422
|
const norm = normalizePatternText(rawText);
|
|
348
423
|
if (!norm) continue;
|
|
424
|
+
// A serialized payload fragment is a transport artifact, not a lesson.
|
|
425
|
+
if (looksLikeSerializedFragment(rawText) || looksLikeSerializedFragment(norm)) continue;
|
|
349
426
|
const words = keywords(norm);
|
|
350
427
|
if (words.length < 2) continue;
|
|
351
428
|
const patKey = words.slice(0, 4).join('_');
|
|
@@ -520,23 +597,60 @@ function isSpecificKeyword(word) {
|
|
|
520
597
|
// Whole-word matching: a bare includes() let "app" hit "apps/", "application" and "happen".
|
|
521
598
|
// Boundaries are non-alphanumerics, so path and punctuation separators still delimit tokens
|
|
522
599
|
// (`src/jobs/queue.js` matches the word "jobs").
|
|
600
|
+
//
|
|
601
|
+
// NVHBM-style optimization (2026-08-28): the previous implementation compiled a NEW RegExp
|
|
602
|
+
// for every word on every call — per-tool-call "memory bandwidth" burned on regex construction
|
|
603
|
+
// instead of matching. The boundary scan below is semantics-identical (same boundary class:
|
|
604
|
+
// non-alphanumeric delimits the word) without any regex compilation, and never throws on
|
|
605
|
+
// punctuation-heavy words.
|
|
606
|
+
function isWordChar(ch) {
|
|
607
|
+
return (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z') || (ch >= '0' && ch <= '9');
|
|
608
|
+
}
|
|
609
|
+
|
|
523
610
|
function containsWholeWord(haystack, word) {
|
|
524
|
-
const
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
611
|
+
const hay = String(haystack == null ? '' : haystack).toLowerCase();
|
|
612
|
+
const w = String(word == null ? '' : word).toLowerCase();
|
|
613
|
+
if (!hay || !w) return false;
|
|
614
|
+
let idx = hay.indexOf(w);
|
|
615
|
+
while (idx !== -1) {
|
|
616
|
+
const before = idx === 0 ? '' : hay[idx - 1];
|
|
617
|
+
const afterIdx = idx + w.length;
|
|
618
|
+
const after = afterIdx >= hay.length ? '' : hay[afterIdx];
|
|
619
|
+
if (!isWordChar(before) && !isWordChar(after)) return true;
|
|
620
|
+
idx = hay.indexOf(w, idx + 1);
|
|
621
|
+
}
|
|
622
|
+
return false;
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
// Token membership is O(1) per word once the haystack is split, but splitting is O(n) —
|
|
626
|
+
// callers that check many word lists against the SAME haystack (the guard loop) must pass
|
|
627
|
+
// the precomputed set instead of letting each check re-split. Compound words (containing
|
|
628
|
+
// '-' or '_') can never appear in the token set (splitting breaks on both), so they fall
|
|
629
|
+
// back to the boundary scan — same rule isSpecificKeyword uses to treat them as strong hits.
|
|
630
|
+
function buildHaystackTokens(normalizedInput) {
|
|
631
|
+
const tokens = new Set();
|
|
632
|
+
for (const token of String(normalizedInput || '').toLowerCase().split(/[^a-z0-9]+/)) {
|
|
633
|
+
if (token) tokens.add(token);
|
|
529
634
|
}
|
|
635
|
+
return tokens;
|
|
530
636
|
}
|
|
531
637
|
|
|
532
|
-
function hasTwoKeywordHits(normalizedInput, words) {
|
|
638
|
+
function hasTwoKeywordHits(normalizedInput, words, precomputedTokens) {
|
|
533
639
|
if (!normalizedInput || !words || words.length === 0) return false;
|
|
640
|
+
const tokens = precomputedTokens || buildHaystackTokens(normalizedInput);
|
|
534
641
|
let hits = 0;
|
|
535
642
|
const seen = new Set();
|
|
536
643
|
for (const word of words) {
|
|
537
644
|
if (!word || seen.has(word)) continue;
|
|
538
645
|
seen.add(word);
|
|
539
|
-
|
|
646
|
+
// Pure-alphanumeric words are O(1) Set lookups against the precomputed token
|
|
647
|
+
// set. Anything with punctuation (compound words with '-'/'_', or any other
|
|
648
|
+
// symbol) can never equal a token, so it keeps the boundary scan — exactly
|
|
649
|
+
// the cases isSpecificKeyword treats as strong single-hit evidence.
|
|
650
|
+
const matched = /^[a-z0-9]+$/i.test(word)
|
|
651
|
+
? tokens.has(word.toLowerCase())
|
|
652
|
+
: containsWholeWord(normalizedInput, word);
|
|
653
|
+
if (!matched) continue;
|
|
540
654
|
// A specific compound/long token carries a match on its own; generic words need two.
|
|
541
655
|
if (isSpecificKeyword(word)) return true;
|
|
542
656
|
hits++;
|
|
@@ -640,6 +754,10 @@ function readGuardArtifact(filePath) {
|
|
|
640
754
|
// evaluateCompiledGuards (fast path)
|
|
641
755
|
// ---------------------------------------------------------------------------
|
|
642
756
|
|
|
757
|
+
// Memoized normalize(guard.text) per guard object. WeakMap keeps artifacts
|
|
758
|
+
// pristine (no added keys that would leak into JSON round-trips or deep-equals).
|
|
759
|
+
const GUARD_NORM_TEXT_CACHE = new WeakMap();
|
|
760
|
+
|
|
643
761
|
/**
|
|
644
762
|
* Check compiled artifact against toolName + toolInput.
|
|
645
763
|
*
|
|
@@ -655,12 +773,23 @@ function evaluateCompiledGuards(artifact, toolName, toolInput) {
|
|
|
655
773
|
|
|
656
774
|
const normInput = normalizeActionText(buildMatchHaystack(toolInput));
|
|
657
775
|
const normTool = (toolName || '').toLowerCase();
|
|
776
|
+
// NVHBM-style: hoist the O(n) haystack tokenization out of the per-guard loop
|
|
777
|
+
// (one split per call, O(1) membership per word). Guard-text normalization is
|
|
778
|
+
// memoized per guard object (WeakMap — no artifact mutation, no JSON leakage)
|
|
779
|
+
// so repeated evaluations of the same in-memory artifact pay it once.
|
|
780
|
+
const haystackTokens = buildHaystackTokens(normInput);
|
|
658
781
|
|
|
659
782
|
for (const guard of artifact.guards) {
|
|
660
|
-
|
|
661
|
-
|
|
783
|
+
if (guard === null || typeof guard !== 'object') continue;
|
|
784
|
+
let normText = GUARD_NORM_TEXT_CACHE.get(guard);
|
|
785
|
+
if (normText === undefined) {
|
|
786
|
+
normText = normalize(guard.text || '');
|
|
787
|
+
GUARD_NORM_TEXT_CACHE.set(guard, normText);
|
|
788
|
+
}
|
|
789
|
+
const toolMentioned = normText.includes(normTool) || normTool === 'unknown';
|
|
662
790
|
|
|
663
|
-
const
|
|
791
|
+
const guardWords = Array.isArray(guard.words) ? guard.words : [];
|
|
792
|
+
const keywordMatch = hasTwoKeywordHits(normInput, guardWords, haystackTokens);
|
|
664
793
|
|
|
665
794
|
// Match if: keyword hits in input, OR tool mentioned + high count.
|
|
666
795
|
// Previously tool-name matching only worked for short inputs — this was
|
|
@@ -697,9 +826,11 @@ function evaluateCompiledGuards(artifact, toolName, toolInput) {
|
|
|
697
826
|
function evaluatePretoolFromState(state, toolName, toolInput) {
|
|
698
827
|
const normInput = normalizeActionText(buildMatchHaystack(toolInput));
|
|
699
828
|
const normTool = (toolName || '').toLowerCase();
|
|
829
|
+
// Same NVHBM-style hoist as the compiled path: tokenize the haystack once.
|
|
830
|
+
const haystackTokens = buildHaystackTokens(normInput);
|
|
700
831
|
|
|
701
832
|
for (const pattern of state.recurringNegativePatterns || []) {
|
|
702
|
-
if (hasTwoKeywordHits(normInput, pattern.words || [])) {
|
|
833
|
+
if (hasTwoKeywordHits(normInput, pattern.words || [], haystackTokens)) {
|
|
703
834
|
const mode = pattern.count >= 3 ? 'block' : 'warn';
|
|
704
835
|
return {
|
|
705
836
|
mode,
|
|
@@ -861,9 +992,11 @@ module.exports = {
|
|
|
861
992
|
inferToolName,
|
|
862
993
|
classify,
|
|
863
994
|
keywords,
|
|
995
|
+
looksLikeSerializedFragment,
|
|
864
996
|
hashText,
|
|
865
997
|
hasTwoKeywordHits,
|
|
866
998
|
buildMatchHaystack,
|
|
999
|
+
buildHaystackTokens,
|
|
867
1000
|
normalizeActionText,
|
|
868
1001
|
isSpecificKeyword,
|
|
869
1002
|
containsWholeWord,
|