thumbgate 1.30.0 → 1.34.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.well-known/mcp/server-card.json +1 -1
  3. package/README.md +54 -16
  4. package/adapters/claude/.mcp.json +2 -2
  5. package/adapters/forge/forge.yaml +3 -3
  6. package/adapters/mcp/server-stdio.js +105 -10
  7. package/adapters/opencode/opencode.json +1 -1
  8. package/bench/observability-eval-suite.json +2 -2
  9. package/bin/cli.js +168 -31
  10. package/config/evals/generation-quality-golden.json +95 -0
  11. package/config/evals/rag-answer-quality-golden.json +91 -0
  12. package/config/evals/retrieval-hybrid-ablation.json +66 -0
  13. package/config/evals/retrieval-ranking-golden.json +522 -0
  14. package/config/gates/claim-verifiers.example.json +42 -0
  15. package/config/gates/claim-verifiers.json +25 -0
  16. package/config/gates/default.json +217 -50
  17. package/config/mcp-allowlists.json +233 -206
  18. package/config/model-tiers.json +7 -2
  19. package/glama.json +6 -0
  20. package/hooks/hooks.json +1 -1
  21. package/package.json +69 -12
  22. package/public/assets/diagrams/before-after.svg +17 -16
  23. package/public/assets/diagrams/hero-thumbs.svg +68 -0
  24. package/public/assets/diagrams/loop.svg +19 -13
  25. package/public/assets/diagrams/self-improving-thumbs-loop.svg +105 -0
  26. package/public/compare.html +1 -0
  27. package/public/dashboard.html +126 -28
  28. package/public/evaluations.html +1 -1
  29. package/public/index.html +142 -13
  30. package/public/numbers.html +3 -2
  31. package/public/pricing.html +143 -30
  32. package/scripts/a-plus-evidence-scorecard.js +303 -0
  33. package/scripts/agent-readiness.js +110 -0
  34. package/scripts/async-eval-observability.js +36 -11
  35. package/scripts/audit-trail.js +37 -1
  36. package/scripts/auto-promote-gates.js +149 -34
  37. package/scripts/auto-wire-hooks.js +20 -8
  38. package/scripts/cli-schema.js +14 -0
  39. package/scripts/colbert-style-maxsim.js +236 -0
  40. package/scripts/cross-encoder-reranker.js +356 -126
  41. package/scripts/dashboard-chat.js +350 -17
  42. package/scripts/document-intake.js +283 -7
  43. package/scripts/eval-quality-suite.js +204 -0
  44. package/scripts/feedback-loop.js +115 -7
  45. package/scripts/feedback-paths.js +32 -13
  46. package/scripts/feedback-quality.js +53 -0
  47. package/scripts/feedback-schema.js +3 -0
  48. package/scripts/file-ledger-lock.js +130 -0
  49. package/scripts/filesystem-search.js +17 -7
  50. package/scripts/financial-control-plane.js +1514 -0
  51. package/scripts/gates-engine.js +202 -7
  52. package/scripts/gemini-embedding-policy.js +1 -0
  53. package/scripts/harness-tool-names.js +70 -0
  54. package/scripts/hook-runtime.js +15 -3
  55. package/scripts/hook-stop-anti-claim.js +63 -3
  56. package/scripts/human-escalation.js +353 -41
  57. package/scripts/lesson-db.js +16 -5
  58. package/scripts/lesson-embedding-index.js +67 -20
  59. package/scripts/lesson-embedding-maintenance.js +177 -0
  60. package/scripts/lesson-reranker.js +55 -9
  61. package/scripts/lesson-retrieval.js +305 -29
  62. package/scripts/lesson-search.js +22 -8
  63. package/scripts/llm-client.js +304 -15
  64. package/scripts/model-tier-router.js +593 -0
  65. package/scripts/pragmatic-hybrid-search.js +379 -0
  66. package/scripts/provider-action-normalizer.js +11 -4
  67. package/scripts/rag-document-pipeline.js +461 -0
  68. package/scripts/rag-structured-output.js +441 -0
  69. package/scripts/ragas-style-metrics.js +351 -0
  70. package/scripts/request-envelope.js +178 -0
  71. package/scripts/rerank-pipeline.js +370 -0
  72. package/scripts/rerank-quality-eval.js +155 -0
  73. package/scripts/retrieval-hybrid-ablation.js +120 -0
  74. package/scripts/retrieval-quality-tier.js +118 -0
  75. package/scripts/secret-scanner.js +395 -4
  76. package/scripts/self-distill-agent.js +7 -1
  77. package/scripts/self-healing-check.js +25 -0
  78. package/scripts/skill-packs.js +183 -0
  79. package/scripts/slow-loop.js +72 -0
  80. package/scripts/statusline-links.js +1 -1
  81. package/scripts/statusline.sh +8 -1
  82. package/scripts/telemetry-analytics.js +13 -1
  83. package/scripts/thumbgate-search.js +98 -6
  84. package/scripts/tier-budget-guard.js +186 -0
  85. package/scripts/tool-registry.js +141 -5
  86. package/scripts/universal-claim-evaluator.js +767 -0
  87. package/scripts/vector-store.js +154 -17
  88. package/scripts/verify-marketing-pages-deployed.js +85 -3
  89. package/scripts/workflow-sentinel.js +77 -11
  90. package/server.json +44 -0
  91. package/smithery.yaml +17 -0
  92. package/src/api/server.js +196 -13
@@ -4,7 +4,7 @@
4
4
  <meta charset="utf-8">
5
5
  <meta name="viewport" content="width=device-width, initial-scale=1">
6
6
  <title>Pricing — Enforcement that pays for itself | ThumbGate</title>
7
- <meta name="description" content="ThumbGate vs DIY prompts, platform hire, or GRC suite. Free evaluate, Pro $19/mo, $499 Diagnostic gate. Countable units that convert.">
7
+ <meta name="description" content="Choose ThumbGate by the value and scope of the agent risk: free evaluation, Pro for one operator, a $499 managed diagnostic for one workflow, or custom enterprise intake.">
8
8
  <link rel="canonical" href="__APP_ORIGIN__/pricing">
9
9
  <link rel="alternate" type="text/markdown" title="ThumbGate LLM context" href="__APP_ORIGIN__/llm-context.md">
10
10
  <meta property="og:title" content="ThumbGate Pricing — $499 Diagnostic · Pro $19/mo">
@@ -57,6 +57,19 @@
57
57
  .fit-card { border-top:3px solid var(--green); background:var(--card); padding:22px; }
58
58
  .fit-card h3 { margin:0 0 8px; }
59
59
  .fit-card p { margin:0; color:var(--muted); }
60
+ .value-equation { display:grid; grid-template-columns:1fr auto 1fr auto 1fr auto 1.3fr; gap:12px; align-items:stretch; margin-top:26px; }
61
+ .value-factor { display:grid; align-content:center; min-height:112px; border:1px solid var(--line); border-radius:14px; background:var(--card); padding:18px; }
62
+ .value-factor strong { font-size:1.06rem; }
63
+ .value-factor span { color:var(--muted); font-size:.83rem; }
64
+ .value-operator { display:grid; place-items:center; color:var(--green); font-size:1.7rem; font-weight:900; }
65
+ .value-result { border-color:var(--green); background:#e9f3ee; }
66
+ .segment-grid { display:grid; grid-template-columns:repeat(3,1fr); gap:14px; margin-top:28px; }
67
+ .segment-card { display:flex; flex-direction:column; border:1px solid var(--line); border-radius:14px; background:var(--card); padding:22px; }
68
+ .segment-card.recommended { border:2px solid var(--green); }
69
+ .segment-card h3 { margin:8px 0 6px; }
70
+ .segment-card p { margin:0 0 14px; color:var(--muted); font-size:.92rem; }
71
+ .segment-card .fence { min-height:58px; margin-top:auto; color:var(--ink); font-size:.82rem; }
72
+ .segment-card .button { margin-top:14px; }
60
73
  .faq { border-top:1px solid var(--line); }
61
74
  .faq-item { border-bottom:1px solid var(--line); }
62
75
  .faq-q { width:100%; border:0; background:transparent; padding:18px 0; color:var(--ink); font:inherit; font-weight:800; text-align:left; cursor:pointer; }
@@ -65,22 +78,52 @@
65
78
  footer { border-top:1px solid var(--line); padding:28px 0; color:var(--muted); font-size:.85rem; }
66
79
  .footer-inner { display:flex; justify-content:space-between; gap:24px; }
67
80
  .footer-links { display:flex; gap:18px; }
68
- @media (max-width:820px) { .hero,.steps,.fit { grid-template-columns:1fr; } .hero { gap:34px; } .nav-links a:not(.nav-buy) { display:none; } }
81
+ @media (max-width:820px) { .hero,.steps,.fit,.segment-grid { grid-template-columns:1fr; } .value-equation { grid-template-columns:1fr; } .value-operator { min-height:24px; } .hero { gap:34px; } .nav-links a:not(.nav-buy) { display:none; } }
69
82
  </style>
70
83
  <script type="application/ld+json">
71
84
  {
72
85
  "@context": "https://schema.org",
73
- "@type": "Service",
74
- "name": "ThumbGate Enterprise Workflow Gate",
75
- "description": "A bounded implementation for one supported local AI-agent workflow: review, configured pre-action gate, regression test, and rollout and rollback proof.",
76
- "provider": { "@type": "Organization", "name": "ThumbGate", "url": "__APP_ORIGIN__/" },
77
- "offers": {
78
- "@type": "Offer",
79
- "price": "499",
80
- "priceCurrency": "USD",
81
- "url": "__APP_ORIGIN__/pricing#buy",
82
- "availability": "https://schema.org/InStock"
83
- }
86
+ "@graph": [
87
+ {
88
+ "@type": "Service",
89
+ "name": "ThumbGate Managed Workflow Diagnostic",
90
+ "description": "A bounded implementation for one supported local AI-agent workflow: review, configured pre-action gate, regression test, and rollout and rollback proof.",
91
+ "provider": { "@type": "Organization", "name": "ThumbGate", "url": "__APP_ORIGIN__/" },
92
+ "audience": { "@type": "BusinessAudience", "audienceType": "Engineering and platform teams with a repeated AI-agent failure" },
93
+ "offers": {
94
+ "@type": "Offer",
95
+ "name": "Managed workflow diagnostic",
96
+ "price": "499",
97
+ "priceCurrency": "USD",
98
+ "url": "__APP_ORIGIN__/pricing#buy",
99
+ "availability": "https://schema.org/InStock"
100
+ }
101
+ },
102
+ {
103
+ "@type": "SoftwareApplication",
104
+ "name": "ThumbGate Pro",
105
+ "applicationCategory": "DeveloperApplication",
106
+ "operatingSystem": "macOS, Linux, Windows",
107
+ "offers": [
108
+ {
109
+ "@type": "Offer",
110
+ "name": "ThumbGate Pro monthly",
111
+ "price": "19",
112
+ "priceCurrency": "USD",
113
+ "url": "__APP_ORIGIN__/checkout/pro?plan_id=pro",
114
+ "availability": "https://schema.org/InStock"
115
+ },
116
+ {
117
+ "@type": "Offer",
118
+ "name": "ThumbGate Pro annual",
119
+ "price": "149",
120
+ "priceCurrency": "USD",
121
+ "url": "__APP_ORIGIN__/checkout/pro?plan_id=pro",
122
+ "availability": "https://schema.org/InStock"
123
+ }
124
+ ]
125
+ }
126
+ ]
84
127
  }
85
128
  </script>
86
129
  <script type="application/ld+json">
@@ -109,8 +152,8 @@
109
152
  <div class="nav-links">
110
153
  <a href="/guide">Technical setup</a>
111
154
  <a href="https://github.com/IgorGanapolsky/ThumbGate/blob/main/docs/VERIFICATION_EVIDENCE.md" target="_blank" rel="noopener">Proof</a>
112
- <a class="nav-enterprise" href="#enterprise-gate" data-offer-link data-cta-id="pricing_nav_buy">Get Started · $499</a>
113
- <a class="nav-pro" href="/checkout/pro?utm_source=pricing&amp;utm_medium=nav&amp;utm_campaign=rally_style_gtm&amp;cta_id=pricing_nav_pro&amp;plan_id=pro" data-offer-link data-cta-id="pricing_nav_pro">Pro · $19/mo</a>
155
+ <a class="nav-enterprise" href="#enterprise-gate" data-offer-link data-cta-id="pricing_nav_buy" data-plan-id="sprint_diagnostic" data-value="499" data-segment="workflow_team">Get Started · $499</a>
156
+ <a class="nav-pro" href="/checkout/pro?utm_source=pricing&amp;utm_medium=nav&amp;utm_campaign=value_packaging_v1&amp;cta_id=pricing_nav_pro&amp;plan_id=pro" data-offer-link data-cta-id="pricing_nav_pro" data-plan-id="pro" data-value="19" data-segment="solo_operator">Pro · $19/mo</a>
114
157
  </div>
115
158
  </div>
116
159
  </nav>
@@ -139,29 +182,74 @@
139
182
  <input type="hidden" name="plan_id" value="sprint_diagnostic">
140
183
  <input type="hidden" name="utm_source" value="pricing">
141
184
  <input type="hidden" name="utm_medium" value="direct_checkout">
142
- <input type="hidden" name="utm_campaign" value="rally_style_gtm">
185
+ <input type="hidden" name="utm_campaign" value="value_packaging_v1">
186
+ <input id="pricing-buyer-segment" type="hidden" name="campaign_variant" value="workflow_team">
143
187
  <input type="hidden" name="cta_id" value="pricing_managed_gate_buy">
144
188
  <input type="hidden" name="cta_placement" value="pricing">
145
189
  <input type="hidden" name="landing_path" value="/pricing">
146
190
  <button class="button" type="submit">Get Started — $499 Diagnostic</button>
147
191
  </form>
148
192
  <p class="fine">Secure payment. We reply with scheduling and the workflow checklist.</p>
149
- <div class="boundary">Detected secret exfiltration and gate-process bypass attempts deny by default. Other destructive actions warn by default and deny when strict enforcement is enabled.</div>
193
+ <div class="boundary">Detected secret exfiltration denies by default: literal secrets, secret-file paths in uploads/pipes/redirections/command-substitution, secret env vars to network tools, scp/rsync/cloud CLI uploads of credential files, and common interpreter one-liners. Not a full network sandbox. Other destructive actions warn by default and deny when strict enforcement is enabled.</div>
150
194
  </aside>
151
195
  <aside class="pro-card" id="pro" aria-label="ThumbGate Pro">
152
196
  <div class="eyebrow">Self-serve</div>
153
197
  <div class="price">$19<span style="font-size:1rem;font-weight:700">/mo</span></div>
154
198
  <p class="price-note">Or $149/yr. Individual operators. Cancel anytime.</p>
155
- <a class="button" href="/checkout/pro?utm_source=pricing&amp;utm_medium=card&amp;utm_campaign=rally_style_gtm&amp;cta_id=pricing_pro_buy&amp;plan_id=pro" data-offer-link data-cta-id="pricing_pro_buy" style="width:100%;box-sizing:border-box">Start Pro — $19/mo</a>
199
+ <a class="button" href="/checkout/pro?utm_source=pricing&amp;utm_medium=card&amp;utm_campaign=value_packaging_v1&amp;cta_id=pricing_pro_buy&amp;plan_id=pro" data-offer-link data-cta-id="pricing_pro_buy" data-plan-id="pro" data-value="19" data-segment="solo_operator" style="width:100%;box-sizing:border-box">Start Pro</a>
156
200
  <p class="fine">Hosted checkout. Free local evaluate remains free after you subscribe.</p>
157
201
  </aside>
158
202
  </div>
159
203
  </div>
160
204
 
205
+ <section id="value">
206
+ <div class="eyebrow">Price from your exposure</div>
207
+ <h2>Value is the recovery work the next repeat does not create.</h2>
208
+ <p class="section-lede">Use your own incident history. ThumbGate does not claim a universal savings number.</p>
209
+ <div class="value-equation" role="img" aria-label="Repeat incidents multiplied by recovery hours and loaded hourly cost, plus downstream impact, equals avoidable recovery exposure">
210
+ <div class="value-factor"><strong>Repeat incidents</strong><span>per quarter</span></div>
211
+ <div class="value-operator">×</div>
212
+ <div class="value-factor"><strong>Recovery hours</strong><span>diagnosis, repair, review</span></div>
213
+ <div class="value-operator">×</div>
214
+ <div class="value-factor"><strong>Loaded hourly cost</strong><span>the people pulled into recovery</span></div>
215
+ <div class="value-operator">+</div>
216
+ <div class="value-factor value-result"><strong>Downstream impact</strong><span>delay, incident response, audit evidence</span></div>
217
+ </div>
218
+ </section>
219
+
220
+ <section id="segments">
221
+ <div class="eyebrow">Choose by buyer value</div>
222
+ <h2>One product. Three buying decisions.</h2>
223
+ <p class="section-lede">The fence is scope and service level—not an arbitrary discount.</p>
224
+ <div class="segment-grid">
225
+ <article class="segment-card">
226
+ <div class="eyebrow">Solo operator</div>
227
+ <h3>Evaluate free, then Pro</h3>
228
+ <p>Pro removes rule limits and adds personal recall, review, dashboard, and export workflows.</p>
229
+ <div class="fence"><strong>Fence:</strong> one operator, local-first, self-serve support. $19 monthly or $149 annual.</div>
230
+ <a class="button" href="/checkout/pro?utm_source=pricing&amp;utm_medium=segment_card&amp;utm_campaign=value_packaging_v1&amp;cta_id=pricing_segment_pro&amp;plan_id=pro" data-offer-link data-cta-id="pricing_segment_pro" data-plan-id="pro" data-value="19" data-segment="solo_operator">Choose Pro</a>
231
+ </article>
232
+ <article class="segment-card recommended">
233
+ <div class="eyebrow">Engineering or platform team</div>
234
+ <h3>Managed diagnostic</h3>
235
+ <p>Turn one expensive, repeated agent failure into a configured gate and a regression proof.</p>
236
+ <div class="fence"><strong>Fence:</strong> one supported local workflow, fixed scope, two-business-day delivery. $499 once.</div>
237
+ <a class="button" href="#enterprise-gate" data-offer-link data-cta-id="pricing_segment_diagnostic" data-plan-id="sprint_diagnostic" data-value="499" data-segment="workflow_team">Harden one workflow</a>
238
+ </article>
239
+ <article class="segment-card">
240
+ <div class="eyebrow">Regulated or multi-workflow team</div>
241
+ <h3>Enterprise service intake</h3>
242
+ <p>Scope approval boundaries, rollback evidence, and accountable rollout support after one workflow proves fit.</p>
243
+ <div class="fence"><strong>Fence:</strong> custom scope after intake. Hosted team features, SSO, and SIEM are not general availability.</div>
244
+ <a class="button" href="#enterprise-gate" data-offer-link data-cta-id="pricing_segment_enterprise" data-plan-id="enterprise_service" data-value="0" data-segment="regulated_team">Start with one diagnostic</a>
245
+ </article>
246
+ </div>
247
+ </section>
248
+
161
249
  <section>
162
250
  <div class="eyebrow">How ThumbGate stacks up</div>
163
251
  <h2>Make the alternative look expensive.</h2>
164
- <p class="section-lede">Same move as high-converting SaaS pricing: DIY prompts cost nothing and stop nothing; platform hire and GRC suites cost months.</p>
252
+ <p class="section-lede">DIY prompts look cheap but still consume model and review time. Platform hiring and GRC suites require larger commitments.</p>
165
253
  <div style="overflow-x:auto;margin-top:22px;border:1px solid var(--line);border-radius:14px;background:var(--card);">
166
254
  <table style="width:100%;border-collapse:collapse;font-size:.92rem;min-width:620px;">
167
255
  <thead>
@@ -176,24 +264,24 @@
176
264
  <tbody>
177
265
  <tr>
178
266
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);color:var(--muted);">Monthly cost</td>
179
- <td style="padding:12px 14px;border-bottom:1px solid var(--line);">$0</td>
180
- <td style="padding:12px 14px;border-bottom:1px solid var(--line);">$15k+ loaded</td>
181
- <td style="padding:12px 14px;border-bottom:1px solid var(--line);">$$$$</td>
267
+ <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Model + review time</td>
268
+ <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Loaded engineering cost</td>
269
+ <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Contract + implementation</td>
182
270
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);font-weight:800;color:var(--green);">$0–$19 · $499 once</td>
183
271
  </tr>
184
272
  <tr>
185
273
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);color:var(--muted);">Time to first gate</td>
186
- <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Never reliable</td>
274
+ <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Manual upkeep</td>
187
275
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Weeks</td>
188
276
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);">Months</td>
189
277
  <td style="padding:12px 14px;border-bottom:1px solid var(--line);font-weight:800;">Minutes after install / 2 business days managed</td>
190
278
  </tr>
191
279
  <tr>
192
280
  <td style="padding:12px 14px;color:var(--muted);">When agent “looks competent”</td>
193
- <td style="padding:12px 14px;">Ignores the prompt</td>
281
+ <td style="padding:12px 14px;">Prompt may be skipped</td>
194
282
  <td style="padding:12px 14px;">Custom glue</td>
195
283
  <td style="padding:12px 14px;">Gateway only</td>
196
- <td style="padding:12px 14px;font-weight:800;">Hard block at tool call</td>
284
+ <td style="padding:12px 14px;font-weight:800;">Configured gate at tool call</td>
197
285
  </tr>
198
286
  </tbody>
199
287
  </table>
@@ -254,13 +342,31 @@
254
342
  } catch (_) {}
255
343
  }
256
344
 
257
- function offerProps(ctaId, placement) {
258
- return { ctaId: ctaId, ctaPlacement: placement, planId: 'sprint_diagnostic', value: 499, currency: 'USD' };
345
+ function offerProps(ctaId, placement, planId, value, segment) {
346
+ return {
347
+ ctaId: ctaId,
348
+ ctaPlacement: placement,
349
+ planId: planId,
350
+ value: Number(value || 0),
351
+ currency: 'USD',
352
+ segment: segment,
353
+ experimentId: 'value_packaging_v1'
354
+ };
259
355
  }
260
356
 
261
357
  document.querySelectorAll('[data-offer-link]').forEach(function (link) {
262
358
  link.addEventListener('click', function () {
263
- const props = offerProps(link.dataset.ctaId || 'pricing_anchor', 'pricing_anchor');
359
+ if (link.getAttribute('href') === '#enterprise-gate') {
360
+ const buyerSegment = document.querySelector('#pricing-buyer-segment');
361
+ if (buyerSegment) buyerSegment.value = link.dataset.segment || 'workflow_team';
362
+ }
363
+ const props = offerProps(
364
+ link.dataset.ctaId || 'pricing_anchor',
365
+ 'pricing_anchor',
366
+ link.dataset.planId || 'unknown',
367
+ link.dataset.value || '0',
368
+ link.dataset.segment || 'unknown'
369
+ );
264
370
  window.plausible('pricing_cta_click', { props: props });
265
371
  sendFirstPartyTelemetry('cta_click', props);
266
372
  });
@@ -269,7 +375,14 @@
269
375
  const checkoutForm = document.querySelector('[data-primary-checkout]');
270
376
  if (checkoutForm) {
271
377
  checkoutForm.addEventListener('submit', function () {
272
- const props = offerProps('pricing_managed_gate_buy', 'pricing');
378
+ const buyerSegment = document.querySelector('#pricing-buyer-segment');
379
+ const props = offerProps(
380
+ 'pricing_managed_gate_buy',
381
+ 'pricing',
382
+ 'sprint_diagnostic',
383
+ 499,
384
+ buyerSegment ? buyerSegment.value : 'workflow_team'
385
+ );
273
386
  window.plausible('checkout_start', { props: props });
274
387
  sendFirstPartyTelemetry('checkout_start', props);
275
388
  });
@@ -283,7 +396,7 @@
283
396
  });
284
397
  });
285
398
 
286
- sendFirstPartyTelemetry('pricing_page_view', { landingPath: '/pricing' });
399
+ sendFirstPartyTelemetry('pricing_page_view', { landingPath: '/pricing', experimentId: 'value_packaging_v1' });
287
400
  </script>
288
401
  </body>
289
402
  </html>
@@ -0,0 +1,303 @@
1
+ #!/usr/bin/env node
2
+ 'use strict';
3
+
4
+ /**
5
+ * Fail-closed A+ readiness scorecard.
6
+ *
7
+ * A passing unit test or a checked-in module is repository evidence, not live
8
+ * production or commercial proof. This scorecard keeps those surfaces separate
9
+ * and awards A+/10 only when every check in every area is verified.
10
+ */
11
+
12
+ const fs = require('node:fs');
13
+ const path = require('node:path');
14
+
15
+ const ROOT = path.join(__dirname, '..');
16
+ const SCORECARD_VERSION = '2026-08-01.1';
17
+
18
+ function read(root, relativePath) {
19
+ try {
20
+ return fs.readFileSync(path.join(root, relativePath), 'utf8');
21
+ } catch {
22
+ return '';
23
+ }
24
+ }
25
+
26
+ function hasAll(text, needles) {
27
+ return needles.every((needle) => text.includes(needle));
28
+ }
29
+
30
+ function safeEval(fn) {
31
+ try {
32
+ return fn() === true;
33
+ } catch {
34
+ return false;
35
+ }
36
+ }
37
+
38
+ function collectRepositoryEvidence(root = ROOT) {
39
+ const landing = read(root, 'public/index.html');
40
+ const feedback = read(root, 'scripts/feedback-loop.js');
41
+ const promotion = read(root, 'scripts/auto-promote-gates.js');
42
+ const buyerPaths = read(root, 'scripts/buyer-paths.js');
43
+ const hook = read(root, 'scripts/hook-pre-tool-use.js');
44
+ const gatesEngine = read(root, 'scripts/gates-engine.js');
45
+ const gateEvasionMatrix = read(root, 'tests/gate-evasion-matrix.test.js');
46
+ const retrieval = read(root, 'scripts/lesson-retrieval.js');
47
+ const crossEncoder = read(root, 'scripts/cross-encoder-reranker.js');
48
+ const productionArchitecture = read(root, 'docs/RAG_PRODUCTION_ARCHITECTURE.md');
49
+ const packageJson = read(root, 'package.json');
50
+
51
+ let qualitySuitePassed = false;
52
+ let rerankGoldenPassed = false;
53
+ if (root === ROOT) {
54
+ qualitySuitePassed = safeEval(() => require('./eval-quality-suite').runSuite().report.passed);
55
+ rerankGoldenPassed = safeEval(() => require('./rerank-quality-eval').evaluate().pass);
56
+ }
57
+
58
+ return {
59
+ landingVisualLoop: hasAll(landing, [
60
+ 'hero-thumbs',
61
+ 'before-after.svg',
62
+ 'self-improving-thumbs-loop.svg',
63
+ 'Is it really self-improving?',
64
+ ]),
65
+ landingBuyerRoutes: hasAll(landing, ['/checkout/pro', '/go/diagnostic-pay'])
66
+ && hasAll(buyerPaths, ['/go/pro', '/go/sprint', '/diagnostic']),
67
+ feedbackRewardReachable: feedback.includes('scoreFeedbackReward('),
68
+ feedbackPromotionReachable: promotion.includes('promote') && hook.includes('retrieveWithRerankingSync'),
69
+ preventionChangePromoted: feedback.includes('whatToChange')
70
+ && feedback.includes('promotion'),
71
+ architectureNamesHonest: hasAll(productionArchitecture, [
72
+ 'LLM-as-a-judge output is diagnostic',
73
+ 'A heuristic score is never',
74
+ 'does not get to override a hard gate',
75
+ ]),
76
+ deterministicMultiQuery: retrieval.includes('buildQueryVariants'),
77
+ hydeExplicitAndBounded: hasAll(retrieval, ['hydeGenerator', 'hydeApplied', 'hydeProvider']),
78
+ rerankProductionWired: crossEncoder.includes("require('./rerank-pipeline')")
79
+ && crossEncoder.includes('rerankPipelineSync(query, candidates'),
80
+ rerankProvenance: hasAll(crossEncoder, [
81
+ 'pairwiseHeuristicScore',
82
+ 'crossEncoderScore',
83
+ 'reranker',
84
+ ]),
85
+ rerankGoldenPassed,
86
+ qualitySuitePassed,
87
+ requestEnvelope: fs.existsSync(path.join(root, 'scripts/request-envelope.js')),
88
+ hardBudgets: fs.existsSync(path.join(root, 'scripts/tier-budget-guard.js')),
89
+ degradedRetrievalFlags: fs.existsSync(path.join(root, 'scripts/retrieval-quality-tier.js')),
90
+ structuredOutputValidation: fs.existsSync(path.join(root, 'scripts/rag-structured-output.js')),
91
+ tenantAclBeforeRetrieval: hasAll(productionArchitecture, [
92
+ 'Filtering must happen before',
93
+ 'Missing/mismatched tenant or principal',
94
+ ]),
95
+ commandPositionHardening: hasAll(gatesEngine, [
96
+ 'LITERAL_COMMAND_SUBSTITUTION_HEADS',
97
+ 'canonicalizeLiteralCommandSubstitutionHead',
98
+ ]) && gateEvasionMatrix.includes('literal command substitution'),
99
+ rawFrameworkDecisionDefended: hasAll(productionArchitecture, [
100
+ '## Framework decision',
101
+ 'LangChain',
102
+ 'LangGraph',
103
+ 'LlamaIndex',
104
+ '## One complete RAG request',
105
+ ]),
106
+ scorecardInMainTest: packageJson.includes('test:a-plus-evidence'),
107
+ };
108
+ }
109
+
110
+ function check(id, label, passed, evidenceClass, remediation) {
111
+ return {
112
+ id,
113
+ label,
114
+ passed: passed === true,
115
+ evidenceClass,
116
+ remediation: passed === true ? null : remediation,
117
+ };
118
+ }
119
+
120
+ function shaMatches(live = {}) {
121
+ const candidate = String(live.candidateBuildSha || '').trim();
122
+ const deployed = String(live.deployedBuildSha || '').trim();
123
+ return candidate.length >= 7 && deployed.length >= 7 && candidate === deployed;
124
+ }
125
+
126
+ function evaluateReadiness({ repo = {}, live = {} } = {}) {
127
+ const production = live.production || {};
128
+ const retrieval = live.retrieval || {};
129
+ const security = live.security || {};
130
+ const commercial = live.commercial || {};
131
+
132
+ const areas = [
133
+ {
134
+ id: 'landing_conversion',
135
+ label: 'Landing page and conversion',
136
+ checks: [
137
+ check('visual_loop', 'Thumb visuals and simple learning diagrams ship', repo.landingVisualLoop, 'repository', 'Ship the visual thumbs-to-gate loop.'),
138
+ check('buyer_routes', 'First-party buyer routes are present', repo.landingBuyerRoutes, 'repository', 'Restore diagnostic, Pro, and sprint buyer routes.'),
139
+ check('live_landing', 'Candidate landing page is verified live', production.landingVerified === true && shaMatches(production), 'production', 'Verify the exact candidate SHA on the live landing page.'),
140
+ ],
141
+ },
142
+ {
143
+ id: 'self_improvement',
144
+ label: 'Self-improving control loop',
145
+ checks: [
146
+ check('reward_reachable', 'Feedback reward scoring is invoked by capture', repo.feedbackRewardReachable, 'repository', 'Wire reward scoring into the capture path.'),
147
+ check('promotion_reachable', 'Reviewed failures reach promotion and pre-action retrieval', repo.feedbackPromotionReachable, 'repository', 'Connect feedback promotion to the pre-action hook.'),
148
+ check('specific_change', 'Specific what-to-change guidance reaches prevention rules', repo.preventionChangePromoted, 'repository', 'Promote specific corrective instructions, not vague signals.'),
149
+ check('live_feedback', 'Fresh production feedback closes the loop', production.feedbackLoopVerified === true, 'production', 'Capture one real reviewed outcome and prove its next-action effect.'),
150
+ ],
151
+ },
152
+ {
153
+ id: 'architecture_honesty',
154
+ label: 'Judge, routing, and architecture honesty',
155
+ checks: [
156
+ check('honest_names', 'Judge, heuristic, neural, and enforcement stages are distinct', repo.architectureNamesHonest, 'repository', 'Document stage placement and prevent misleading model-level MoE claims.'),
157
+ check('route_trace', 'Live traces identify the provider and routed model', production.providerTraceVerified === true, 'production', 'Attach a secret-safe live route trace.'),
158
+ ],
159
+ },
160
+ {
161
+ id: 'query_transformation',
162
+ label: 'Query transformation, multi-query, and HyDE',
163
+ checks: [
164
+ check('multi_query', 'Bounded deterministic multi-query is implemented', repo.deterministicMultiQuery, 'repository', 'Implement bounded, inspectable query variants.'),
165
+ check('hyde_contract', 'HyDE is explicit, bounded, and provenance-bearing', repo.hydeExplicitAndBounded, 'repository', 'Add an explicit caller-supplied HyDE contract and fallback.'),
166
+ check('hyde_holdout', 'HyDE or multi-query improves a provider holdout', retrieval.queryTransformationHoldoutPassed === true, 'provider-holdout', 'Measure lift on a non-fixture provider holdout.'),
167
+ ],
168
+ },
169
+ {
170
+ id: 'reranking',
171
+ label: 'Reranking cascade',
172
+ checks: [
173
+ check('production_wiring', 'BM25F, local MaxSim, and pairwise fusion run in PreToolUse', repo.rerankProductionWired, 'repository', 'Wire the documented cascade into the production caller.'),
174
+ check('provenance', 'Heuristic and neural scores cannot masquerade as each other', repo.rerankProvenance, 'repository', 'Emit per-stage provenance and explicit fallbacks.'),
175
+ check('golden', 'Deterministic rerank golden floors pass', repo.rerankGoldenPassed, 'deterministic-eval', 'Fix rerank golden regressions.'),
176
+ check('neural_holdout', 'True neural pair/late-interaction holdout passes', retrieval.neuralRerankHoldoutPassed === true, 'provider-holdout', 'Run a pretrained pair scorer or token embedder on an external holdout.'),
177
+ check('llm_failures', 'LLM rerank failure modes pass live-provider tests', retrieval.llmRerankFailureModesPassed === true, 'provider-holdout', 'Test malformed, partial, injected, timed-out, and unavailable LLM reranks.'),
178
+ ],
179
+ },
180
+ {
181
+ id: 'evaluation',
182
+ label: 'Retrieval and answer evaluation',
183
+ checks: [
184
+ check('offline_suite', 'Recall, precision, MRR, nDCG, and answer proxy floors pass', repo.qualitySuitePassed, 'deterministic-eval', 'Fix the unified deterministic quality suite.'),
185
+ check('external_cases', 'External labeled holdout has at least 100 cases', Number(retrieval.externalHoldoutCases) >= 100, 'provider-holdout', 'Label and freeze at least 100 non-fixture cases.'),
186
+ check('judge_calibration', 'LLM judge is calibrated against human labels', retrieval.judgeCalibrationPassed === true, 'provider-holdout', 'Measure judge agreement and calibration against human-reviewed labels.'),
187
+ ],
188
+ },
189
+ {
190
+ id: 'production_controls',
191
+ label: 'Latency, cost, and observability',
192
+ checks: [
193
+ check('request_envelope', 'Request trace, token, cost, and retrieval envelope exists', repo.requestEnvelope, 'repository', 'Add a request envelope.'),
194
+ check('hard_budgets', 'Per-request and daily tier budgets fail closed', repo.hardBudgets, 'repository', 'Add hard cost and tier budgets.'),
195
+ check('degraded_flags', 'Stale or stub retrieval is labeled degraded', repo.degradedRetrievalFlags, 'repository', 'Expose retrieval quality tiers.'),
196
+ check('live_slo', 'Production p95 and cost SLOs pass under load', production.loadTestPassed === true && Number(production.p95LatencyMs) > 0, 'production', 'Run a production-like load test and attach p95/cost evidence.'),
197
+ check('cache_batch', 'Live cache and batching savings are measured', production.cacheAndBatchingMeasured === true, 'production', 'Measure cache hit rate and batching cost/latency lift.'),
198
+ ],
199
+ },
200
+ {
201
+ id: 'failure_security',
202
+ label: 'Failure modes, validation, ACL, and tenancy',
203
+ checks: [
204
+ check('structured', 'Structured answers and citations are validated', repo.structuredOutputValidation, 'repository', 'Validate output shape and citation relationships.'),
205
+ check('acl_order', 'Tenant/document ACL runs before retrieval and hydration', repo.tenantAclBeforeRetrieval, 'repository', 'Enforce authorization before ranking.'),
206
+ check('command_evasion', 'Literal command-substitution evasions are canonicalized and tested', repo.commandPositionHardening, 'repository', 'Ratchet deterministic command-position substitutions in the evasion matrix.'),
207
+ check('penetration_test', 'Tenant isolation has external penetration evidence', security.tenantPenTestPassed === true, 'security-review', 'Run a professional tenant-isolation penetration test.'),
208
+ check('incident_drill', 'Hallucination, stale-index, miss, and leak drills pass', security.failureDrillPassed === true, 'production', 'Run and retain production-like failure drills.'),
209
+ ],
210
+ },
211
+ {
212
+ id: 'framework_pipeline',
213
+ label: 'Framework decision and end-to-end RAG defense',
214
+ checks: [
215
+ check('decision', 'Raw versus LangChain/LangGraph/LlamaIndex tradeoffs are defended', repo.rawFrameworkDecisionDefended, 'repository', 'Document the complete pipeline and framework decision.'),
216
+ check('ratchet', 'The evidence scorecard runs in the main test chain', repo.scorecardInMainTest, 'repository', 'Wire this scorecard into the test chain.'),
217
+ ],
218
+ },
219
+ {
220
+ id: 'commercial_validation',
221
+ label: 'Value, willingness to pay, and captured money',
222
+ checks: [
223
+ check('buyer_conversations', 'At least 10 target-buyer value conversations are evidenced', Number(commercial.buyerConversations) >= 10, 'commercial', 'Complete and retain 10 target-buyer value conversations.'),
224
+ check('payment_asks', 'At least 3 exact-price payment asks are evidenced', Number(commercial.paymentAsks) >= 3, 'commercial', 'Make three exact-price payment asks to qualified buyers.'),
225
+ check('external_payment', 'At least one non-owner external payment is reconciled', Number(commercial.externalPayments) >= 1, 'provider', 'Capture and reconcile one real external payment.'),
226
+ check('provider_truth', 'Provider catalog and product attribution are verified', commercial.providerRevenueVerified === true, 'provider', 'Attach exact provider catalog and product-attributed revenue evidence.'),
227
+ ],
228
+ },
229
+ ];
230
+
231
+ for (const area of areas) {
232
+ const passed = area.checks.filter((row) => row.passed).length;
233
+ area.score = Number((10 * passed / area.checks.length).toFixed(1));
234
+ area.grade = area.score === 10 ? 'A+' : area.score >= 9 ? 'A' : area.score >= 8 ? 'B' : area.score >= 7 ? 'C' : area.score >= 6 ? 'D' : 'F';
235
+ area.status = area.score === 10 ? 'verified' : passed === 0 ? 'blocked' : 'partial';
236
+ }
237
+
238
+ const score = Number((areas.reduce((sum, area) => sum + area.score, 0) / areas.length).toFixed(1));
239
+ const atTarget = areas.every((area) => area.score === 10);
240
+ return {
241
+ scorecardVersion: SCORECARD_VERSION,
242
+ generatedAt: new Date().toISOString(),
243
+ target: { score: 10, grade: 'A+' },
244
+ atTarget,
245
+ score,
246
+ grade: atTarget ? 'A+' : score >= 9 ? 'A' : score >= 8 ? 'B' : score >= 7 ? 'C' : score >= 6 ? 'D' : 'F',
247
+ areas,
248
+ blockers: areas.flatMap((area) => area.checks
249
+ .filter((row) => !row.passed)
250
+ .map((row) => ({ area: area.id, check: row.id, evidenceClass: row.evidenceClass, remediation: row.remediation }))),
251
+ };
252
+ }
253
+
254
+ function formatMarkdown(report) {
255
+ const lines = [
256
+ '# ThumbGate A+ evidence scorecard',
257
+ '',
258
+ `Overall: **${report.score}/10 (${report.grade})**`,
259
+ `Target verified: **${report.atTarget ? 'YES' : 'NO'}**`,
260
+ '',
261
+ '| Area | Score | Grade | Status |',
262
+ '|---|---:|:---:|---|',
263
+ ...report.areas.map((area) => `| ${area.label} | ${area.score}/10 | ${area.grade} | ${area.status} |`),
264
+ '',
265
+ '## Remaining evidence blockers',
266
+ '',
267
+ ...(report.blockers.length
268
+ ? report.blockers.map((row) => `- **${row.area}/${row.check}** (${row.evidenceClass}): ${row.remediation}`)
269
+ : ['- None. Every repository, production, provider, security, and commercial check is verified.']),
270
+ '',
271
+ ];
272
+ return lines.join('\n');
273
+ }
274
+
275
+ function loadLiveEvidence(argv) {
276
+ const index = argv.indexOf('--evidence');
277
+ if (index === -1 || !argv[index + 1]) return {};
278
+ return JSON.parse(fs.readFileSync(path.resolve(argv[index + 1]), 'utf8'));
279
+ }
280
+
281
+ function main() {
282
+ const live = loadLiveEvidence(process.argv.slice(2));
283
+ const repo = collectRepositoryEvidence();
284
+ const report = evaluateReadiness({ repo, live });
285
+ process.stdout.write(process.argv.includes('--json')
286
+ ? `${JSON.stringify(report, null, 2)}\n`
287
+ : `${formatMarkdown(report)}\n`);
288
+ if (process.argv.includes('--require-a-plus') && !report.atTarget) process.exitCode = 1;
289
+ }
290
+
291
+ function isCliEntrypoint(argv = process.argv) {
292
+ return Boolean(argv[1]) && path.resolve(argv[1]) === path.resolve(__filename);
293
+ }
294
+
295
+ if (isCliEntrypoint()) main();
296
+
297
+ module.exports = {
298
+ SCORECARD_VERSION,
299
+ collectRepositoryEvidence,
300
+ evaluateReadiness,
301
+ formatMarkdown,
302
+ isCliEntrypoint,
303
+ };