thumbgate 1.35.0 → 1.37.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/cobble-hot-store-compare-not-clone/SKILL.md +80 -0
- package/.agents/skills/colab-compute-honesty-not-clone/SKILL.md +70 -0
- package/.agents/skills/cyberstrike-compare-not-clone/SKILL.md +36 -0
- package/.agents/skills/gitlab-sandbox-allowlist-not-trust/SKILL.md +77 -0
- package/.agents/skills/jit-harness-compare-not-clone/SKILL.md +34 -0
- package/.agents/skills/openui-catalog-compose-honesty/SKILL.md +64 -0
- package/.agents/skills/token-shunt-honesty-not-clone/SKILL.md +67 -0
- package/.agents/skills/typesafe-typed-questions-not-clone/SKILL.md +82 -0
- package/.agents/skills/zvec-grep-compare-not-clone/SKILL.md +34 -0
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/llms.txt +1 -0
- package/.well-known/mcp/server-card.json +1 -1
- package/CONTRIBUTING.md +95 -0
- package/README.md +195 -632
- package/THIRD_PARTY_NOTICES.md +89 -0
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/future-agi/.mcp.json +8 -0
- package/adapters/future-agi/FUTURE_AGI.md +23 -0
- package/adapters/future-agi/config.toml +3 -0
- package/adapters/future-agi/future-agi-bridge.js +9 -0
- package/adapters/future-agi/opencode.json +8 -0
- package/adapters/herdr/herdr-plugin.toml +18 -0
- package/adapters/mcp/server-stdio.js +238 -25
- package/adapters/opencode/opencode.json +1 -1
- package/adapters/workos/WORKOS.md +52 -0
- package/bin/cli.js +465 -3
- package/bin/futureagi-bridge +9 -0
- package/config/gate-templates.json +629 -4
- package/config/gates/actor-critic-audit.json +34 -0
- package/config/gates/default.json +9 -3
- package/config/gates/five-walls-governance.json +34 -0
- package/config/gates/future-agi-guardrails.json +34 -0
- package/config/gates/radware-threat-defense-2026.json +61 -0
- package/config/gates/simatree-data-governance.json +33 -0
- package/config/mcp-allowlists.json +4 -0
- package/config/merge-quality-checks.json +6 -0
- package/config/model-candidates.json +312 -29
- package/config/model-tiers.json +18 -0
- package/config/post-deploy-marketing-pages.json +10 -0
- package/config/progressive/01-wire-only.json +11 -0
- package/config/progressive/02-dashboard-empty-ok.json +10 -0
- package/config/progressive/03-one-lesson.json +10 -0
- package/config/progressive/04-warn-fires.json +11 -0
- package/config/progressive/05-strict-optional.json +11 -0
- package/config/progressive/README.md +15 -0
- package/config/schemas/broker-execution-receipt.schema.json +139 -0
- package/config/schemas/provider-execution-attestation-v1.schema.json +58 -0
- package/conformance/provider-attestation/vectors.json +320 -0
- package/docs/specs/provider-execution-attestation-v1.md +69 -0
- package/openapi/openapi.yaml +15 -0
- package/package.json +408 -150
- package/public/about.html +2 -2
- package/public/ai-malpractice-prevention.html +7 -7
- package/public/blog/a-10-dollar-vps-is-not-a-computer.html +143 -0
- package/public/blog/a-receipt-is-not-world-state.html +388 -0
- package/public/blog/git-at-agent-scale.html +374 -0
- package/public/blog/no-llm-in-the-gate.html +133 -0
- package/public/blog.html +80 -0
- package/public/case-studies.html +16 -1
- package/public/compare.html +28 -0
- package/public/diagnostic.html +216 -7
- package/public/docs/connectors.html +39 -0
- package/public/federal.html +2 -2
- package/public/founders.html +639 -0
- package/public/index.html +87 -9
- package/public/install.html +8 -8
- package/public/learn.html +39 -0
- package/public/numbers.html +2 -2
- package/public/peter.html +310 -0
- package/public/platform-partners.html +119 -0
- package/public/pricing.html +24 -3
- package/public/privacy.html +117 -0
- package/public/pro.html +17 -0
- package/public/support.html +62 -0
- package/public/terms.html +130 -0
- package/public/third-party-notices.html +95 -0
- package/public/yt.html +351 -0
- package/scripts/action-receipts.js +133 -3
- package/scripts/adaptive-governance-arena.js +349 -0
- package/scripts/admin-override.js +205 -0
- package/scripts/agent-action-inventory.js +869 -0
- package/scripts/agent-audit-trace.js +42 -2
- package/scripts/agent-egress-policy.js +1117 -0
- package/scripts/agent-memory-lifecycle.js +141 -2
- package/scripts/agent-operations-planner.js +441 -1
- package/scripts/agent-readiness.js +68 -0
- package/scripts/agent-security-central.js +647 -0
- package/scripts/allowlist-bridge-honesty.js +417 -0
- package/scripts/async-job-runner.js +102 -11
- package/scripts/audit-trail.js +212 -0
- package/scripts/billing.js +1 -1
- package/scripts/broker-execution-receipts.js +719 -0
- package/scripts/budget-aware-gates-proof.js +423 -0
- package/scripts/claude-feedback-sync.js +29 -3
- package/scripts/claw-harness-production.js +237 -0
- package/scripts/cli-schema.js +203 -1
- package/scripts/cobble-hot-store-split.js +600 -0
- package/scripts/codex-runbook-flywheel.js +318 -0
- package/scripts/colab-compute-honesty.js +281 -0
- package/scripts/context-footprint.js +186 -0
- package/scripts/contextfs.js +143 -61
- package/scripts/dashboard.js +251 -32
- package/scripts/deepseek-v4-runtime-guardrails.js +72 -6
- package/scripts/docker-sandbox-planner.js +18 -0
- package/scripts/double-blind-eval-protocol.js +252 -0
- package/scripts/edotenv-rl-gateway.js +259 -0
- package/scripts/ensure-production-search-corpus.js +162 -0
- package/scripts/eval-holdout.js +311 -0
- package/scripts/feedback-aggregate.js +21 -2
- package/scripts/feedback-loop.js +87 -5
- package/scripts/feedback-quality.js +9 -0
- package/scripts/file-ledger-lock.js +4 -1
- package/scripts/financial-control-plane.js +41 -1
- package/scripts/find-dormant-requires.js +118 -0
- package/scripts/fs-utils.js +84 -8
- package/scripts/gates-engine.js +826 -68
- package/scripts/generate-case-study-outreach.js +24 -15
- package/scripts/git-at-scale.js +628 -0
- package/scripts/governance-conflict-audit.js +1650 -0
- package/scripts/governance-difficulty-curriculum.js +328 -0
- package/scripts/graphrag-retrieval.js +275 -0
- package/scripts/gurobi-optimizer.js +324 -0
- package/scripts/gurobi_optimizer.py +485 -0
- package/scripts/harness-selector.js +82 -1
- package/scripts/hidden-entry-points.js +284 -0
- package/scripts/human-escalation.js +199 -1
- package/scripts/hybrid-feedback-context.js +152 -19
- package/scripts/intent-governed-execution.js +602 -0
- package/scripts/intervention-policy.js +123 -20
- package/scripts/jit-harness-compose.js +628 -0
- package/scripts/jsonl-watcher.js +10 -0
- package/scripts/lesson-embedding-index.js +95 -12
- package/scripts/lesson-retrieval.js +105 -19
- package/scripts/local-model-profile.js +19 -2
- package/scripts/mailer/resend-mailer.js +1 -1
- package/scripts/matryoshka-embedding.js +235 -0
- package/scripts/mcp-oauth.js +42 -4
- package/scripts/mcp-session-handles.js +1016 -0
- package/scripts/mcp-wiring-doctor.js +314 -0
- package/scripts/memory-firewall.js +115 -2
- package/scripts/memory-scope-readiness.js +299 -0
- package/scripts/memory-vs-rag-route.js +161 -0
- package/scripts/model-tier-router.js +148 -21
- package/scripts/nvidia-specdecode-al-doctor.js +536 -0
- package/scripts/openui-catalog-compose-honesty.js +593 -0
- package/scripts/operational-integrity.js +19 -1
- package/scripts/override-audit.js +213 -0
- package/scripts/package-manager-honesty-doctor.js +458 -0
- package/scripts/pr-manager.js +63 -1
- package/scripts/prove-herdr-adapter.js +52 -0
- package/scripts/prove-memory-pyramid-and-symbolic-canvas.js +95 -0
- package/scripts/prove-workos.js +73 -0
- package/scripts/provider-attestation-conformance.js +192 -0
- package/scripts/provider-receipt-contract.js +136 -0
- package/scripts/qwen38-max-cost-optimizer.js +401 -0
- package/scripts/radware-threat-defense.js +280 -0
- package/scripts/rag-embedding-identity.js +221 -0
- package/scripts/rag-precision-guardrails.js +112 -2
- package/scripts/remote-feedback-capture.js +159 -0
- package/scripts/research-agent-harness.js +256 -0
- package/scripts/rsi-safety-hillclimb.js +200 -0
- package/scripts/rule-sprawl.js +188 -0
- package/scripts/schedule-manager.js +147 -0
- package/scripts/self-heal.js +8 -0
- package/scripts/session-lease.js +415 -0
- package/scripts/simatree-data-governance.js +347 -0
- package/scripts/slo-alert-engine.js +172 -7
- package/scripts/solver-parity.js +539 -0
- package/scripts/stealth-memory-injection-gate.js +333 -0
- package/scripts/switchyard-router.js +366 -0
- package/scripts/telemetry-analytics.js +84 -27
- package/scripts/temporal-decay-weighting.js +138 -0
- package/scripts/test-all.js +165 -0
- package/scripts/token-savings.js +42 -0
- package/scripts/token-shunt-honesty.js +502 -0
- package/scripts/tool-kpi-tracker.js +108 -5
- package/scripts/tool-registry.js +193 -5
- package/scripts/typesafe-typed-questions.js +813 -0
- package/scripts/universal-claim-evaluator.js +14 -2
- package/scripts/vector-store.js +279 -9
- package/scripts/workflow-notebook.js +391 -0
- package/scripts/workflow-sentinel.js +111 -12
- package/scripts/workos-production-guard.js +260 -0
- package/scripts/workspace-search-route.js +515 -0
- package/server.json +2 -2
- package/src/agent-identity-boundary.js +76 -0
- package/src/agent-retrieval-cache.js +155 -0
- package/src/alert-noise-ledger.js +502 -0
- package/src/api/server.js +724 -153
- package/src/git-fast-cache.js +220 -0
- package/src/git-wal-sync.js +156 -0
- package/src/hash-anchored-edit.js +82 -0
- package/src/hermes-platform-protocol.js +475 -0
- package/src/hermes-sync-plane.js +241 -0
- package/src/index.js +30 -1
- package/src/iso42001-compliance-guard.js +97 -0
- package/src/latency-budget.js +244 -0
- package/src/mcp-writeguard.js +316 -0
- package/src/miminions-adapter.js +106 -0
- package/src/pipeline-compass.js +104 -0
- package/src/ppl-alert-pipeline.js +284 -0
- package/src/rendezvous-router.js +90 -0
- package/src/security-questionnaire.js +195 -0
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Codex Runbook Flywheel — ThumbGate steal of OpenAI's "Automating repetitive
|
|
5
|
+
* work at OpenAI with Codex" workflow (developers.openai.com blog).
|
|
6
|
+
*
|
|
7
|
+
* The episode's loop, verbatim in spirit:
|
|
8
|
+
* run the workflow -> document it (commands, results, dead ends,
|
|
9
|
+
* decisions) -> the next run reuses what earlier runs learned.
|
|
10
|
+
*
|
|
11
|
+
* Four enforcement primitives, all deterministic:
|
|
12
|
+
*
|
|
13
|
+
* 1. Plan-before-act. Codex writes the plan into the notebook and WAITS
|
|
14
|
+
* for approval before executing. -> newRunbook() starts at 'plan';
|
|
15
|
+
* execute() refuses to run without an approved plan.
|
|
16
|
+
*
|
|
17
|
+
* 2. Consequential choices need human judgment. Automatic approval review
|
|
18
|
+
* handles eligible actions WITHOUT changing permission boundaries.
|
|
19
|
+
* -> autoReview() marks only bounded, reversible actions eligible;
|
|
20
|
+
* consequential ones stay on the human queue.
|
|
21
|
+
*
|
|
22
|
+
* 3. Capture decisions that would otherwise vanish into chat history:
|
|
23
|
+
* which option was chosen, why, what to do differently next time.
|
|
24
|
+
* -> captureDecision() appends to a per-workflow decision log.
|
|
25
|
+
*
|
|
26
|
+
* 4. Cheap discovery for the next run: a companion index over past
|
|
27
|
+
* runbooks (the *.index.md analog), searchable by workflow name.
|
|
28
|
+
* -> buildIndex() / discoverContext().
|
|
29
|
+
*/
|
|
30
|
+
|
|
31
|
+
const RUN_STATES = Object.freeze(['plan', 'approved', 'running', 'done', 'blocked']);
|
|
32
|
+
|
|
33
|
+
const CONSEQUENTIAL = Object.freeze([
|
|
34
|
+
'payment', 'external-email', 'production-deploy', 'delete', 'permission-change', 'publish',
|
|
35
|
+
]);
|
|
36
|
+
|
|
37
|
+
const ALLOWED_ACTIONS = Object.freeze([
|
|
38
|
+
'read-file', 'list-files', 'grep', 'search', 'read-notes',
|
|
39
|
+
'list-models', 'estimate-cost', 'check-quota',
|
|
40
|
+
]);
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Create a runbook. Starts at the plan stage — execution is impossible until
|
|
44
|
+
* a human approves. Mirrors "wait for me to review and approve the plan."
|
|
45
|
+
*/
|
|
46
|
+
function newRunbook(workflow, goal) {
|
|
47
|
+
if (!workflow || !goal) {
|
|
48
|
+
throw new Error('runbook needs a workflow name and a goal');
|
|
49
|
+
}
|
|
50
|
+
return {
|
|
51
|
+
id: require('crypto').randomUUID(),
|
|
52
|
+
workflow,
|
|
53
|
+
goal,
|
|
54
|
+
state: 'plan',
|
|
55
|
+
plan: [],
|
|
56
|
+
decisions: [],
|
|
57
|
+
steps: [],
|
|
58
|
+
deadEnds: [],
|
|
59
|
+
createdAt: new Date().toISOString(),
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Normalize a step to its stable identifier. String steps are used directly.
|
|
65
|
+
* Object steps must have an `id` property; otherwise approval is rejected.
|
|
66
|
+
* This prevents the approved-snapshot and executeStep from disagreeing due
|
|
67
|
+
* to object reference identity.
|
|
68
|
+
*/
|
|
69
|
+
function stepId(step) {
|
|
70
|
+
if (typeof step === 'string') return step.trim() || null;
|
|
71
|
+
if (step && typeof step === 'object' && typeof step.id === 'string' && step.id.trim()) {
|
|
72
|
+
return step.id.trim();
|
|
73
|
+
}
|
|
74
|
+
return null;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Record the plan and approve it. Approval is an explicit human act.
|
|
79
|
+
* Steps are normalized to immutable identifiers so that executeStep can
|
|
80
|
+
* match against the approved plan without relying on object reference identity.
|
|
81
|
+
*/
|
|
82
|
+
function approvePlan(runbook, plan, approver) {
|
|
83
|
+
if (runbook.state !== 'plan') {
|
|
84
|
+
return { ok: false, reason: `cannot approve from state "${runbook.state}"` };
|
|
85
|
+
}
|
|
86
|
+
if (!Array.isArray(plan) || plan.length === 0) {
|
|
87
|
+
return { ok: false, reason: 'an empty plan cannot be approved' };
|
|
88
|
+
}
|
|
89
|
+
if (!approver || typeof approver !== 'string' || !approver.trim()) {
|
|
90
|
+
return { ok: false, reason: 'approval requires a non-empty, trimmed string approver' };
|
|
91
|
+
}
|
|
92
|
+
// Normalize each step to a stable identifier. Object entries without a
|
|
93
|
+
// usable `id` are rejected rather than silently cloned by reference.
|
|
94
|
+
const normalized = [];
|
|
95
|
+
for (const s of plan) {
|
|
96
|
+
const id = stepId(s);
|
|
97
|
+
if (id === null) {
|
|
98
|
+
return { ok: false, reason: `step ${JSON.stringify(s)} has no stable id — reject or assign one` };
|
|
99
|
+
}
|
|
100
|
+
normalized.push(id);
|
|
101
|
+
}
|
|
102
|
+
runbook.approvedBy = approver.trim();
|
|
103
|
+
runbook.approvedPlan = Object.freeze(normalized);
|
|
104
|
+
runbook.state = 'approved';
|
|
105
|
+
runbook.approvedAt = new Date().toISOString();
|
|
106
|
+
return { ok: true, state: runbook.state };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Automatic approval review for individual actions. Uses an explicit
|
|
111
|
+
* allowlist of safe action types — fail closed for anything unlisted.
|
|
112
|
+
*/
|
|
113
|
+
function autoReview(action) {
|
|
114
|
+
if (!action || !action.type || typeof action.type !== 'string') {
|
|
115
|
+
return { eligible: false, reason: 'missing or invalid action type — human approval required' };
|
|
116
|
+
}
|
|
117
|
+
if (!ALLOWED_ACTIONS.includes(action.type)) {
|
|
118
|
+
return { eligible: false, reason: `"${action.type}" is unlisted — human approval required` };
|
|
119
|
+
}
|
|
120
|
+
if (CONSEQUENTIAL.includes(action.type)) {
|
|
121
|
+
return { eligible: false, reason: `"${action.type}" is consequential — human approval required` };
|
|
122
|
+
}
|
|
123
|
+
if (action.irreversible) {
|
|
124
|
+
return { eligible: false, reason: 'irreversible actions are never auto-approved' };
|
|
125
|
+
}
|
|
126
|
+
return { eligible: true, reason: 'bounded and reversible — auto-review passes' };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Execute one step. Refuses unless the plan was approved.
|
|
131
|
+
*/
|
|
132
|
+
function executeStep(runbook, step, outcome) {
|
|
133
|
+
if (runbook.state !== 'approved' && runbook.state !== 'running') {
|
|
134
|
+
return { ok: false, reason: `execution refused — runbook state is "${runbook.state}", not approved` };
|
|
135
|
+
}
|
|
136
|
+
if (!Array.isArray(runbook.approvedPlan)) {
|
|
137
|
+
return { ok: false, reason: 'execution refused — no approved plan snapshot on record' };
|
|
138
|
+
}
|
|
139
|
+
if (!runbook.approvedPlan.includes(stepId(step))) {
|
|
140
|
+
return { ok: false, reason: `execution refused — "${step}" is not in the approved plan` };
|
|
141
|
+
}
|
|
142
|
+
runbook.state = 'running';
|
|
143
|
+
runbook.steps.push({ step, outcome: outcome || 'ok', at: new Date().toISOString() });
|
|
144
|
+
return { ok: true, executed: runbook.steps.length };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Capture a decision that would otherwise disappear into chat history.
|
|
149
|
+
*/
|
|
150
|
+
function captureDecision(runbook, decision) {
|
|
151
|
+
if (!decision || !decision.choice || !decision.reason) {
|
|
152
|
+
return { ok: false, reason: 'a decision needs both a choice and a reason' };
|
|
153
|
+
}
|
|
154
|
+
runbook.decisions.push({
|
|
155
|
+
choice: decision.choice,
|
|
156
|
+
reason: decision.reason,
|
|
157
|
+
nextTime: decision.nextTime || null,
|
|
158
|
+
at: new Date().toISOString(),
|
|
159
|
+
});
|
|
160
|
+
return { ok: true, decisions: runbook.decisions.length };
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Mark a dead end — the article documents these on purpose so the next run
|
|
165
|
+
* doesn't repeat them.
|
|
166
|
+
*/
|
|
167
|
+
function recordDeadEnd(runbook, deadEnd) {
|
|
168
|
+
runbook.deadEnds.push({ deadEnd, at: new Date().toISOString() });
|
|
169
|
+
return { deadEnds: runbook.deadEnds.length };
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Close the runbook. Only a running/approved runbook can close, and only
|
|
174
|
+
* with at least one recorded step.
|
|
175
|
+
*/
|
|
176
|
+
function closeRunbook(runbook) {
|
|
177
|
+
if (runbook.state !== 'approved' && runbook.state !== 'running') {
|
|
178
|
+
return { ok: false, reason: `cannot close from state "${runbook.state}"` };
|
|
179
|
+
}
|
|
180
|
+
if (runbook.steps.length === 0) {
|
|
181
|
+
return { ok: false, reason: 'cannot close a runbook with no recorded steps' };
|
|
182
|
+
}
|
|
183
|
+
runbook.state = 'done';
|
|
184
|
+
runbook.completedAt = new Date().toISOString();
|
|
185
|
+
return { ok: true, state: 'done' };
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Build the companion index over completed runbooks (the *.index.md analog).
|
|
190
|
+
* Full decision and dead-end records are stored alongside counts so later
|
|
191
|
+
* runs can reuse the captured guidance.
|
|
192
|
+
*/
|
|
193
|
+
function buildIndex(runbooks) {
|
|
194
|
+
return (runbooks || [])
|
|
195
|
+
.filter((r) => r.state === 'done')
|
|
196
|
+
.map((r) => ({
|
|
197
|
+
workflow: r.workflow,
|
|
198
|
+
goal: r.goal,
|
|
199
|
+
steps: r.steps.length,
|
|
200
|
+
decisions: r.decisions.length,
|
|
201
|
+
deadEnds: r.deadEnds.length,
|
|
202
|
+
decisionsLog: r.decisions.map((d) => ({ ...d })),
|
|
203
|
+
deadEndsLog: r.deadEnds.map((d) => ({ ...d })),
|
|
204
|
+
completedAt: r.completedAt,
|
|
205
|
+
}));
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Discover prior context for a workflow — what earlier runs learned.
|
|
210
|
+
* Returns the full decision and dead-end records so callers can act on
|
|
211
|
+
* the captured guidance, not just counts.
|
|
212
|
+
*/
|
|
213
|
+
function discoverContext(index, workflow) {
|
|
214
|
+
const prior = (index || []).filter((e) => e.workflow === workflow);
|
|
215
|
+
return {
|
|
216
|
+
workflow,
|
|
217
|
+
priorRuns: prior.length,
|
|
218
|
+
totalDecisions: prior.reduce((n, e) => n + e.decisions, 0),
|
|
219
|
+
totalDeadEnds: prior.reduce((n, e) => n + e.deadEnds, 0),
|
|
220
|
+
decisions: prior.flatMap((e) => (e.decisionsLog || []).map((d) => ({ ...d }))),
|
|
221
|
+
deadEnds: prior.flatMap((e) => (e.deadEndsLog || []).map((d) => ({ ...d }))),
|
|
222
|
+
reusable: prior.length > 0,
|
|
223
|
+
};
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
function isCliEntrypoint() {
|
|
227
|
+
return require.main === module;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function parseCliArgs(argv) {
|
|
231
|
+
const args = argv.slice(2);
|
|
232
|
+
const mode = { dryRun: false, solve: false };
|
|
233
|
+
for (const arg of args) {
|
|
234
|
+
if (arg === '--dry-run') mode.dryRun = true;
|
|
235
|
+
else if (arg === '--solve') mode.solve = true;
|
|
236
|
+
}
|
|
237
|
+
return mode;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
function main() {
|
|
241
|
+
const { dryRun, solve } = parseCliArgs(process.argv);
|
|
242
|
+
const rb = newRunbook('model-eval', 'Run the evaluation against the current model');
|
|
243
|
+
|
|
244
|
+
const premature = executeStep(rb, 'run eval'); // must refuse
|
|
245
|
+
approvePlan(rb, ['review previous run', 'write plan', 'run eval', 'document'], 'igor');
|
|
246
|
+
const review = { auto: autoReview({ type: 'read-file' }), blocked: autoReview({ type: 'production-deploy' }) };
|
|
247
|
+
|
|
248
|
+
if (dryRun) {
|
|
249
|
+
// Dry-run mode: do not record execution steps or complete the runbook.
|
|
250
|
+
// Surface the plan and auto-review decisions only.
|
|
251
|
+
process.stdout.write(JSON.stringify({
|
|
252
|
+
mode: 'dry-run',
|
|
253
|
+
honesty: 'deterministic model of the OpenAI Codex+Runme runbook flywheel',
|
|
254
|
+
source: 'https://developers.openai.com/blog/automating-repetitive-work-at-openai-with-codex',
|
|
255
|
+
prematureExecution: premature,
|
|
256
|
+
autoReview: review,
|
|
257
|
+
approvedPlan: rb.approvedPlan,
|
|
258
|
+
state: rb.state,
|
|
259
|
+
}, null, 2) + '\n');
|
|
260
|
+
return;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
if (solve) {
|
|
264
|
+
// Solve mode: execute the approved plan end-to-end.
|
|
265
|
+
return runSolve(rb, review, premature);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// Default mode: execute the approved plan end-to-end.
|
|
269
|
+
runSolve(rb, review, premature);
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
function runSolve(rb, review, premature) {
|
|
273
|
+
executeStep(rb, 'review previous run', 'ok');
|
|
274
|
+
executeStep(rb, 'run eval', 'ok');
|
|
275
|
+
captureDecision(rb, {
|
|
276
|
+
choice: 'reuse existing eval cluster',
|
|
277
|
+
reason: 'quota exhausted on new provisioning',
|
|
278
|
+
nextTime: 'check quota before provisioning',
|
|
279
|
+
});
|
|
280
|
+
recordDeadEnd(rb, 'new cluster provisioning — quota exhausted');
|
|
281
|
+
closeRunbook(rb);
|
|
282
|
+
|
|
283
|
+
const index = buildIndex([rb]);
|
|
284
|
+
const context = discoverContext(index, 'model-eval');
|
|
285
|
+
|
|
286
|
+
process.stdout.write(JSON.stringify({
|
|
287
|
+
mode: 'solve',
|
|
288
|
+
honesty: 'deterministic model of the OpenAI Codex+Runme runbook flywheel',
|
|
289
|
+
source: 'https://developers.openai.com/blog/automating-repetitive-work-at-openai-with-codex',
|
|
290
|
+
prematureExecution: premature,
|
|
291
|
+
autoReview: review,
|
|
292
|
+
finalState: rb.state,
|
|
293
|
+
decisions: rb.decisions.length,
|
|
294
|
+
deadEnds: rb.deadEnds.length,
|
|
295
|
+
decisionsLog: rb.decisions,
|
|
296
|
+
deadEndsLog: rb.deadEnds,
|
|
297
|
+
index, context,
|
|
298
|
+
}, null, 2) + '\n');
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
if (isCliEntrypoint()) main();
|
|
302
|
+
|
|
303
|
+
module.exports = {
|
|
304
|
+
RUN_STATES,
|
|
305
|
+
CONSEQUENTIAL,
|
|
306
|
+
ALLOWED_ACTIONS,
|
|
307
|
+
stepId,
|
|
308
|
+
newRunbook,
|
|
309
|
+
approvePlan,
|
|
310
|
+
autoReview,
|
|
311
|
+
executeStep,
|
|
312
|
+
captureDecision,
|
|
313
|
+
recordDeadEnd,
|
|
314
|
+
closeRunbook,
|
|
315
|
+
buildIndex,
|
|
316
|
+
discoverContext,
|
|
317
|
+
isCliEntrypoint,
|
|
318
|
+
};
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Colab signup FORMAT steal — not a Colab clone, not a GPU SKU.
|
|
6
|
+
*
|
|
7
|
+
* Live source 2026-09-17 (BrowserOS, iganapolsky@gmail.com):
|
|
8
|
+
* https://colab.research.google.com/signup
|
|
9
|
+
* Pay As You Go $9.99/100 CU and $49.99/500 CU (buttons disabled on this account)
|
|
10
|
+
* Colab Pro $9.99/mo and Pro+ $49.99/mo still show Subscribe
|
|
11
|
+
* "See current plan" is Google AI Plans — not proof of Pro+
|
|
12
|
+
*
|
|
13
|
+
* Transfers:
|
|
14
|
+
* 1. Free exists; paid buys Compute Units, not a dedicated GPU
|
|
15
|
+
* 2. Subscribe-button visible ≠ already subscribed
|
|
16
|
+
* 3. Background / 24h execution is a Pro+ receipt, not a default
|
|
17
|
+
*
|
|
18
|
+
* Does not buy Colab Pro/Pro+/PAYG. Does not install colab-cli.
|
|
19
|
+
* ThumbGate evals stay on GitHub Actions.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
const path = require('node:path');
|
|
23
|
+
|
|
24
|
+
const SOURCE_URL = 'https://colab.research.google.com/signup';
|
|
25
|
+
|
|
26
|
+
const PLANS = Object.freeze({
|
|
27
|
+
free: { monthlyUsd: 0, includedCu: 0, backgroundHours: 0 },
|
|
28
|
+
payg: { monthlyUsd: 0, includedCu: 0, backgroundHours: 0, packs: [[9.99, 100], [49.99, 500]] },
|
|
29
|
+
pro: { monthlyUsd: 9.99, includedCu: 100, backgroundHours: 0 },
|
|
30
|
+
proplus: { monthlyUsd: 49.99, includedCu: 600, backgroundHours: 24 },
|
|
31
|
+
enterprise: { monthlyUsd: null, includedCu: null, backgroundHours: null },
|
|
32
|
+
unknown: { monthlyUsd: null, includedCu: null, backgroundHours: 0 },
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
const CLONE_RE = /\b(colab-cli|google-colab-pro-runner|ngrok.*colab|colab.*ssh|zero-cost a100|free a100)\b/i;
|
|
36
|
+
const PAID_FEATURE_RE = /\b(pro\+|proplus|a100|v100|background execution|24\s*hours?|high-?ram|premium gpu|compute units?)\b/i;
|
|
37
|
+
const BUY_RE = /\b(buy|subscribe|purchase).{0,40}(colab pro|pro\+|compute units?)\b/i;
|
|
38
|
+
|
|
39
|
+
const RAIL_MAP = Object.freeze([
|
|
40
|
+
{ colab: 'Free always exists; paid is extra CU', thumbgate: 'Do not claim hosted GPU/A100 as the product' },
|
|
41
|
+
{ colab: 'Subscribe button visible ≠ already subscribed', thumbgate: 'Require --plan-proof before Pro/Pro+ claims' },
|
|
42
|
+
{ colab: 'CU packs ($9.99/100, $49.99/500) expire; CU ≠ dedicated GPU', thumbgate: 'Hours remaining need hardware class + CU rate; else fail closed' },
|
|
43
|
+
{ colab: 'Background 24h is Pro+', thumbgate: 'GitHub Actions remains the eval runner; no Colab offload SKU' },
|
|
44
|
+
]);
|
|
45
|
+
|
|
46
|
+
function normalizeBoolean(value) {
|
|
47
|
+
if (value === true || value === 1) return true;
|
|
48
|
+
if (value === false || value === 0 || value == null) return false;
|
|
49
|
+
return /^(1|true|yes|on)$/i.test(String(value).trim());
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function parseSnapshot(raw) {
|
|
53
|
+
if (!raw) return null;
|
|
54
|
+
if (typeof raw === 'object') return raw;
|
|
55
|
+
try {
|
|
56
|
+
return JSON.parse(String(raw));
|
|
57
|
+
} catch {
|
|
58
|
+
return { parseError: true };
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function normalizeOptions(raw = {}) {
|
|
63
|
+
const plan = String(raw.plan || 'unknown').toLowerCase().replace('pro+', 'proplus');
|
|
64
|
+
return {
|
|
65
|
+
claim: String(raw.claim || ''),
|
|
66
|
+
plan: Object.prototype.hasOwnProperty.call(PLANS, plan) ? plan : 'unknown',
|
|
67
|
+
planProof: raw['plan-proof'] || raw.planProof || null,
|
|
68
|
+
snapshot: parseSnapshot(raw.snapshot || raw['signup-snapshot']),
|
|
69
|
+
cloneColab: normalizeBoolean(raw['clone-colab'] || raw.cloneColab),
|
|
70
|
+
buyPro: normalizeBoolean(raw['buy-pro'] || raw.buyPro),
|
|
71
|
+
mapOnly: normalizeBoolean(raw['map-only'] || raw.mapOnly),
|
|
72
|
+
claimReady: normalizeBoolean(raw['claim-ready'] || raw.claimReady),
|
|
73
|
+
json: normalizeBoolean(raw.json),
|
|
74
|
+
strict: normalizeBoolean(raw.strict),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function buildColabComputeHonestyReport(rawOptions = {}) {
|
|
79
|
+
const options = normalizeOptions(rawOptions);
|
|
80
|
+
const findings = [];
|
|
81
|
+
const claim = options.claim;
|
|
82
|
+
const snapshot = options.snapshot && !options.snapshot.parseError ? options.snapshot : null;
|
|
83
|
+
|
|
84
|
+
if (options.snapshot && options.snapshot.parseError) {
|
|
85
|
+
findings.push({
|
|
86
|
+
id: 'snapshot_parse_error',
|
|
87
|
+
severity: 'fail',
|
|
88
|
+
gateId: 'require-compute-unit-proof',
|
|
89
|
+
message: 'Signup snapshot JSON did not parse.',
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (options.cloneColab || CLONE_RE.test(claim)) {
|
|
94
|
+
findings.push({
|
|
95
|
+
id: 'colab_sku_clone',
|
|
96
|
+
severity: 'fail',
|
|
97
|
+
gateId: 'refuse-colab-sku-clone',
|
|
98
|
+
message: 'Refused Colab clone (colab-cli, ngrok/SSH, zero-cost A100, or a Colab-runner SKU).',
|
|
99
|
+
});
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
if (options.buyPro || BUY_RE.test(claim)) {
|
|
103
|
+
findings.push({
|
|
104
|
+
id: 'colab_spend_refused',
|
|
105
|
+
severity: 'fail',
|
|
106
|
+
gateId: 'refuse-colab-sku-clone',
|
|
107
|
+
message: 'Refused buying Colab Pro/Pro+/Compute Units from this doctor. No surprise spend.',
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const wantsPaid = PAID_FEATURE_RE.test(claim);
|
|
112
|
+
const proof = options.planProof ? String(options.planProof).toLowerCase().replace('pro+', 'proplus') : null;
|
|
113
|
+
if (wantsPaid && !options.mapOnly) {
|
|
114
|
+
if (!proof || proof === 'unknown' || proof === 'free') {
|
|
115
|
+
findings.push({
|
|
116
|
+
id: 'paid_feature_without_plan_proof',
|
|
117
|
+
severity: 'fail',
|
|
118
|
+
gateId: 'require-compute-unit-proof',
|
|
119
|
+
message: 'Paid Colab features (Pro/Pro+/A100/CU/24h background) need --plan-proof from a live Current-plan receipt, not a Subscribe button.',
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
if (snapshot && snapshot.proSubscribeVisible && (proof === 'pro' || proof === 'proplus' || /pro\+|proplus/i.test(claim))) {
|
|
125
|
+
findings.push({
|
|
126
|
+
id: 'subscribe_button_is_not_receipt',
|
|
127
|
+
severity: 'fail',
|
|
128
|
+
gateId: 'require-compute-unit-proof',
|
|
129
|
+
message: 'Live signup still shows Subscribe on Pro/Pro+. That is not proof the account is subscribed.',
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (/24\s*hour|background execution/i.test(claim) && proof !== 'proplus') {
|
|
134
|
+
findings.push({
|
|
135
|
+
id: 'background_requires_proplus',
|
|
136
|
+
severity: 'fail',
|
|
137
|
+
gateId: 'require-compute-unit-proof',
|
|
138
|
+
message: '24h background execution is a Pro+ receipt on the signup page, not Free/Pro/PAYG.',
|
|
139
|
+
});
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
if (options.claimReady && findings.some((f) => f.severity === 'fail')) {
|
|
143
|
+
findings.push({
|
|
144
|
+
id: 'claim_without_compute_proof',
|
|
145
|
+
severity: 'fail',
|
|
146
|
+
gateId: 'require-compute-unit-proof',
|
|
147
|
+
message: 'Claimed Colab compute ready without plan proof.',
|
|
148
|
+
});
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const seen = new Set();
|
|
152
|
+
const deduped = [];
|
|
153
|
+
for (const f of findings) {
|
|
154
|
+
if (seen.has(f.id)) continue;
|
|
155
|
+
seen.add(f.id);
|
|
156
|
+
deduped.push(f);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const failCount = deduped.filter((f) => f.severity === 'fail').length;
|
|
160
|
+
let status = 'ready';
|
|
161
|
+
if (failCount > 0) status = 'fail';
|
|
162
|
+
|
|
163
|
+
return {
|
|
164
|
+
name: 'thumbgate-colab-compute-honesty',
|
|
165
|
+
ok: status !== 'fail',
|
|
166
|
+
status,
|
|
167
|
+
source: SOURCE_URL,
|
|
168
|
+
disclaimer:
|
|
169
|
+
'FORMAT steal from Colab paid signup (CU packs, subscribe≠receipt, 24h background is Pro+). Not affiliated with Google Colab. Does not buy a plan or clone a notebook GPU SKU.',
|
|
170
|
+
observed: {
|
|
171
|
+
account: snapshot && snapshot.account,
|
|
172
|
+
paygDisabled: snapshot ? Boolean(snapshot.paygDisabled) : null,
|
|
173
|
+
proSubscribeVisible: snapshot ? Boolean(snapshot.proSubscribeVisible) : null,
|
|
174
|
+
proPlusSubscribeVisible: snapshot ? Boolean(snapshot.proPlusSubscribeVisible) : null,
|
|
175
|
+
},
|
|
176
|
+
plans: PLANS,
|
|
177
|
+
plan: options.plan,
|
|
178
|
+
planProof: proof,
|
|
179
|
+
map: options.mapOnly ? RAIL_MAP : undefined,
|
|
180
|
+
findings: deduped,
|
|
181
|
+
summary: { failCount, findingCount: deduped.length },
|
|
182
|
+
recommendedGates: [...new Set(deduped.map((f) => f.gateId).filter(Boolean))],
|
|
183
|
+
nextActions: [
|
|
184
|
+
'Keep ThumbGate evals on GitHub Actions. Do not offload PreToolUse to Colab.',
|
|
185
|
+
'Treat Subscribe on /signup as not-subscribed until a Current-plan receipt exists.',
|
|
186
|
+
'Do not buy Pro/Pro+/CU from an agent session.',
|
|
187
|
+
'Pair with gates require-compute-unit-proof and refuse-colab-sku-clone.',
|
|
188
|
+
],
|
|
189
|
+
exampleCommand: 'npx thumbgate colab-compute-honesty --json --map-only',
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
function formatColabComputeHonestyReport(report) {
|
|
194
|
+
const lines = [
|
|
195
|
+
'',
|
|
196
|
+
'ThumbGate Colab Compute-Honesty Doctor',
|
|
197
|
+
'-'.repeat(48),
|
|
198
|
+
`Status : ${report.status}`,
|
|
199
|
+
`Plan : ${report.plan} proof=${report.planProof || '(none)'}`,
|
|
200
|
+
`Source : ${report.source}`,
|
|
201
|
+
`Findings: ${report.summary.findingCount} (fail=${report.summary.failCount})`,
|
|
202
|
+
];
|
|
203
|
+
if (report.map) {
|
|
204
|
+
lines.push('', 'Rail map:');
|
|
205
|
+
for (const row of report.map) lines.push(` - ${row.colab} → ${row.thumbgate}`);
|
|
206
|
+
}
|
|
207
|
+
if (report.findings.length) {
|
|
208
|
+
lines.push('', 'Findings:');
|
|
209
|
+
for (const f of report.findings) {
|
|
210
|
+
lines.push(` - [${f.severity}] ${f.id}${f.gateId ? ` [${f.gateId}]` : ''}`);
|
|
211
|
+
lines.push(` ${f.message}`);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
lines.push('', 'Next actions:');
|
|
215
|
+
for (const a of report.nextActions) lines.push(` - ${a}`);
|
|
216
|
+
lines.push('', `Example: ${report.exampleCommand}`);
|
|
217
|
+
lines.push(`Note: ${report.disclaimer}`, '');
|
|
218
|
+
return `${lines.join('\n')}\n`;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
function parseCliArgs(argv) {
|
|
222
|
+
const options = {};
|
|
223
|
+
for (const arg of argv) {
|
|
224
|
+
if (arg === '--json') { options.json = true; continue; }
|
|
225
|
+
if (arg === '--strict') { options.strict = true; continue; }
|
|
226
|
+
if (arg === '--map-only') { options['map-only'] = true; continue; }
|
|
227
|
+
if (arg === '--claim-ready') { options['claim-ready'] = true; continue; }
|
|
228
|
+
if (arg === '--clone-colab') { options['clone-colab'] = true; continue; }
|
|
229
|
+
if (arg === '--buy-pro') { options['buy-pro'] = true; continue; }
|
|
230
|
+
if (arg === '--help' || arg === '-h') { options.help = true; continue; }
|
|
231
|
+
const m = /^--([^=]+)(?:=(.*))?$/.exec(arg);
|
|
232
|
+
if (!m) continue;
|
|
233
|
+
options[m[1]] = m[2] === undefined ? true : m[2];
|
|
234
|
+
}
|
|
235
|
+
return options;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
function printHelp() {
|
|
239
|
+
process.stdout.write(`Usage: node scripts/colab-compute-honesty.js [flags]
|
|
240
|
+
|
|
241
|
+
Flags:
|
|
242
|
+
--claim=TEXT Claim to audit
|
|
243
|
+
--plan=free|payg|pro|proplus|unknown
|
|
244
|
+
--plan-proof=PLAN Live Current-plan receipt (not a Subscribe button)
|
|
245
|
+
--snapshot=JSON Live /signup snapshot
|
|
246
|
+
--map-only
|
|
247
|
+
--claim-ready
|
|
248
|
+
--clone-colab Always fail
|
|
249
|
+
--buy-pro Always fail (no surprise spend)
|
|
250
|
+
--json --strict
|
|
251
|
+
|
|
252
|
+
Source: ${SOURCE_URL}
|
|
253
|
+
`);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
function runCli(argv = process.argv.slice(2)) {
|
|
257
|
+
const args = parseCliArgs(argv);
|
|
258
|
+
if (args.help) {
|
|
259
|
+
printHelp();
|
|
260
|
+
return 0;
|
|
261
|
+
}
|
|
262
|
+
const report = buildColabComputeHonestyReport(args);
|
|
263
|
+
if (args.json) process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
264
|
+
else process.stdout.write(formatColabComputeHonestyReport(report));
|
|
265
|
+
if (args.strict && report.status !== 'ready') return 1;
|
|
266
|
+
if (report.status === 'fail') return 1;
|
|
267
|
+
return 0;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
module.exports = {
|
|
271
|
+
SOURCE_URL,
|
|
272
|
+
PLANS,
|
|
273
|
+
RAIL_MAP,
|
|
274
|
+
buildColabComputeHonestyReport,
|
|
275
|
+
formatColabComputeHonestyReport,
|
|
276
|
+
runCli,
|
|
277
|
+
};
|
|
278
|
+
|
|
279
|
+
if (path.resolve(process.argv[1] || '') === path.resolve(__filename)) {
|
|
280
|
+
process.exitCode = runCli();
|
|
281
|
+
}
|