ruvnet-brain 4.3.21 → 4.3.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/bin/install.mjs +275 -60
- package/console/app.js +189 -9
- package/console/index.html +70 -24
- package/console/scope.css +137 -0
- package/console/scope.html +144 -0
- package/console/scope.js +209 -0
- package/console/style.css +26 -0
- package/console/tips.html +1 -0
- package/kb/corpus-release-identity.mjs +239 -0
- package/kb/update-storage-transaction.mjs +20 -3
- package/package.json +9 -2
- package/plugin/.claude-plugin/plugin.json +2 -2
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/checkpoint.md +61 -0
- package/plugin/hooks/codex-hooks.json +64 -1
- package/plugin/hooks/hook-contracts.json +299 -6
- package/plugin/hooks/hooks.json +81 -1
- package/plugin/mcp/server.mjs +23 -0
- package/plugin/scripts/advocacy-catalog.mjs +245 -0
- package/plugin/scripts/advocacy-route.mjs +460 -0
- package/plugin/scripts/continuation-gate.mjs +25 -2
- package/plugin/scripts/continuation-objective.mjs +7 -1
- package/plugin/scripts/continuity-hook-policy.mjs +190 -15
- package/plugin/scripts/coverage-integrity.mjs +7 -0
- package/plugin/scripts/gates.mjs +113 -10
- package/plugin/scripts/grounding-turn-gate.mjs +167 -0
- package/plugin/scripts/grounding-turn-mark.mjs +91 -0
- package/plugin/scripts/hook-shim.mjs +14 -0
- package/plugin/scripts/nightly-scheduler.mjs +37 -4
- package/plugin/scripts/project-progression-checkpoint.mjs +145 -0
- package/plugin/scripts/project-progression-contract.mjs +16 -0
- package/plugin/scripts/project-progression-hook.mjs +3 -0
- package/plugin/scripts/project-progression-producer.mjs +252 -0
- package/plugin/scripts/project-progression-reader.mjs +271 -0
- package/plugin/scripts/project-progression-session-start.mjs +93 -16
- package/plugin/scripts/project-progression-sources.mjs +220 -0
- package/plugin/scripts/project-progression-store.mjs +106 -13
- package/plugin/scripts/ruvnet-gate1-pattern.mjs +29 -0
- package/plugin/scripts/session-snapshot-hook.mjs +115 -7
- package/plugin/scripts/session-start-budget.mjs +59 -0
- package/plugin/scripts/session-start-core.mjs +234 -457
- package/plugin/scripts/session-start-fsutil.mjs +61 -0
- package/plugin/scripts/session-start-health.mjs +64 -0
- package/plugin/scripts/session-start-hook-description.mjs +45 -0
- package/plugin/scripts/session-start-issue-alert.mjs +77 -0
- package/plugin/scripts/session-start-repo-identity.mjs +54 -0
- package/plugin/scripts/session-start-signals.mjs +73 -0
- package/plugin/scripts/session-start-trace.mjs +86 -0
- package/plugin/scripts/session-start-update-plane.mjs +104 -0
- package/plugin/scripts/unprompted-runtime.mjs +32 -2
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +26 -2
- package/plugin/skills/ruvnet-brain/SKILL.md +67 -2
- package/scripts/adr-072-completion.mjs +1 -1
- package/scripts/agentdb-fleet-doctor.mjs +5 -1
- package/scripts/approved-runtime.mjs +197 -0
- package/scripts/brain-novice-50.mjs +16 -1
- package/scripts/brain-score.mjs +23 -5
- package/scripts/build-bundle.mjs +971 -530
- package/scripts/build-concepts.mjs +36 -116
- package/scripts/console-engine.test.mjs +8 -7
- package/scripts/console-runtime-identity.mjs +4 -0
- package/scripts/corpus-aggregates.mjs +94 -77
- package/scripts/corpus-candidate.mjs +475 -222
- package/scripts/corpus-next-seed.mjs +225 -0
- package/scripts/corpus-promotion.mjs +58 -0
- package/scripts/corpus-reconcile.mjs +411 -105
- package/scripts/doc-currency.mjs +16 -1
- package/scripts/dual-host-deliberation.mjs +25 -2
- package/scripts/dual-host-suggest.mjs +17 -1
- package/scripts/falsify.mjs +13 -3
- package/scripts/gist-receipts.mjs +482 -87
- package/scripts/github-health-watch.mjs +12 -2
- package/scripts/handoff-asset.mjs +34 -0
- package/scripts/hook-retirement-check.mjs +8 -1
- package/scripts/host-registry.mjs +1 -1
- package/scripts/ingest-gists.mjs +74 -101
- package/scripts/job-heartbeat.sh +77 -14
- package/scripts/learning-replay-execution.mjs +10 -4
- package/scripts/nightly-gists.sh +27 -13
- package/scripts/nightly-two-run-proof.mjs +1 -1
- package/scripts/nightly-watchdog.mjs +61 -4
- package/scripts/onboarding-console.mjs +364 -28
- package/scripts/oracle/produce-questions.mjs +293 -0
- package/scripts/oracle/producer-hosts.mjs +235 -0
- package/scripts/oracle/repo-recall.mjs +448 -0
- package/scripts/oracle/retrieval-accuracy.mjs +818 -0
- package/scripts/oracle/source-tree.mjs +165 -0
- package/scripts/oracle/source-units.mjs +391 -0
- package/scripts/oracle/spike-run.mjs +98 -0
- package/scripts/oracle/unit-inventory.mjs +141 -0
- package/scripts/oracle/unit-sampling.mjs +128 -0
- package/scripts/oracle/validate-labels.mjs +250 -0
- package/scripts/private-overlay.mjs +248 -0
- package/scripts/product-integrity-contract.mjs +1 -1
- package/scripts/proxy/claude-proxied.sh +6 -0
- package/scripts/proxy/proxy-revert.sh +5 -0
- package/scripts/proxy/proxy-up.sh +6 -0
- package/scripts/proxy/proxy-verify.mjs +4 -0
- package/scripts/public-inputs.mjs +409 -0
- package/scripts/public-verification-inputs.mjs +112 -26
- package/scripts/public-verification-lane.mjs +1 -1
- package/scripts/published-surface-probe.mjs +34 -4
- package/scripts/qe/card-lane-gate.mjs +16 -1
- package/scripts/qe/session-start-gate.mjs +16 -1
- package/scripts/rebuild-gists-from-receipts.mjs +58 -78
- package/scripts/record-lesson.mjs +4 -1
- package/scripts/rehearse-corpus-pipeline.mjs +994 -0
- package/scripts/release-abort-stale.mjs +5 -1
- package/scripts/release-authority.mjs +104 -12
- package/scripts/release-channel-kind.mjs +86 -0
- package/scripts/release-convergence-watchdog.mjs +7 -2
- package/scripts/release-projection.mjs +177 -72
- package/scripts/release-transaction-provider.mjs +47 -10
- package/scripts/release-transaction.mjs +40 -11
- package/scripts/release.mjs +252 -17
- package/scripts/retrieval-canary.mjs +87 -0
- package/scripts/rvf-index-audit.mjs +573 -13
- package/scripts/rvf-wire.mjs +269 -0
- package/scripts/seal-gist-receipt.mjs +65 -0
- package/scripts/selfcheck.mjs +42 -21
- package/scripts/source-coverage.mjs +253 -24
- package/scripts/status-honesty.mjs +25 -0
- package/scripts/sync-census.mjs +0 -0
- package/scripts/sync-version.mjs +2 -0
- package/scripts/trismart.mjs +42 -0
- package/scripts/updater-manifest.mjs +162 -0
- package/scripts/verify-channels.mjs +17 -5
- package/scripts/wired-check.mjs +48 -10
- package/tri-smart-skill/QUICKSTART.md +37 -0
- package/tri-smart-skill/README.md +92 -0
- package/tri-smart-skill/install.cmd +14 -0
- package/tri-smart-skill/install.command +13 -0
- package/tri-smart-skill/install.mjs +51 -0
- package/tri-smart-skill/install.sh +9 -0
- package/tri-smart-skill/tri-smart/SKILL.md +90 -0
- package/tri-smart-skill/tri-smart/evals/evals.json +25 -0
- package/tri-smart-skill/tri-smart/references/protocol.md +25 -0
- package/tri-smart-skill/tri-smart/references/provider-cli.md +18 -0
- package/tri-smart-skill/tri-smart/scripts/review.mjs +154 -0
- package/tri-smart-skill/tri-smart/scripts/setup.mjs +97 -0
- package/tri-smart-skill/tri-smart/scripts/verify-access.mjs +107 -0
- package/scripts/corpus-seed-publish.mjs +0 -110
|
@@ -1,22 +1,61 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
+
// gist-receipts.mjs — the ONE canonical gist capture/render/vector-build path.
|
|
3
|
+
//
|
|
4
|
+
// Step 2 of the corpus-seed/release pipeline consolidation (2026-09-13). Before this, gist "truth"
|
|
5
|
+
// (whether a gist is reconciled, current, and what its passages hash to) was written and/or read in
|
|
6
|
+
// at least 5 disagreeing places: this file, corpus-aggregates.mjs, rebuild-gists-from-receipts.mjs,
|
|
7
|
+
// release-projection.mjs, and build-bundle.mjs. Every one of those now either calls into
|
|
8
|
+
// `buildGistAggregate` (the single producer) or `validateGistReceipt` (the single validator), or has
|
|
9
|
+
// been deleted outright.
|
|
10
|
+
//
|
|
11
|
+
// THE THREE-STAGE PIPELINE:
|
|
12
|
+
// 1. captureGistSources — fetch (or reuse from an exact, body-verified cache) every gist's exact
|
|
13
|
+
// per-file identity and raw UTF-8 body content. Returns an UNBOUND, unsealed CapturedGistSet —
|
|
14
|
+
// internal working state, never written to disk as if it were a receipt.
|
|
15
|
+
// 2. renderGistPassages — the ONE banner/chunk/JSONL-escaping implementation. Pure function of a
|
|
16
|
+
// CapturedGistSet; no I/O, no network.
|
|
17
|
+
// 3. buildGistAggregate — orchestrates capture -> render -> write -> embed -> seal -> validate,
|
|
18
|
+
// atomically (via a fresh stage directory + promoteArtifactSet), and returns a StoreResult. This
|
|
19
|
+
// is the fix for the 2026-09-12 bug where a reconciled receipt was always sealed with
|
|
20
|
+
// passagesSha256:null: passages are now rendered and hashed from the SAME call that seals the
|
|
21
|
+
// receipt, so a receipt can never be sealed unbound.
|
|
22
|
+
//
|
|
23
|
+
// `validateGistReceipt` is the one shared validator for capture-cache reuse, producer output, and
|
|
24
|
+
// archive verification. It wraps (never reimplements) `validateGistAggregateReceipt` from
|
|
25
|
+
// plugin/scripts/coverage-integrity.mjs, which every other consumer of this receipt schema already
|
|
26
|
+
// uses (source-coverage.mjs, corpus-candidate.mjs) — so the byte-binding contract lives in one place.
|
|
27
|
+
|
|
2
28
|
import crypto from 'node:crypto';
|
|
29
|
+
import fs from 'node:fs';
|
|
3
30
|
import path from 'node:path';
|
|
4
31
|
import { spawnSync } from 'node:child_process';
|
|
5
|
-
import {
|
|
32
|
+
import { fileURLToPath } from 'node:url';
|
|
33
|
+
import { canonicalJson, digest, validateGistAggregateReceipt } from './coverage-integrity.mjs';
|
|
34
|
+
import { promoteArtifactSet } from '../kb/incremental-refresh.mjs';
|
|
35
|
+
import { writeRvfGeneration } from './rvf-generation.mjs';
|
|
6
36
|
|
|
37
|
+
const DEFAULT_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
7
38
|
const TEXT_EXT = new Set(['.md', '.markdown', '.txt', '.rst']);
|
|
39
|
+
const EMBED_MODEL = 'Xenova/bge-base-en-v1.5';
|
|
40
|
+
const EMBED_DIMENSIONS = 768;
|
|
8
41
|
const sha256 = (value) => crypto.createHash('sha256').update(value).digest('hex');
|
|
9
42
|
const HEX40 = /^[a-f0-9]{40}$/;
|
|
10
43
|
const HEX64 = /^[a-f0-9]{64}$/;
|
|
44
|
+
const HEX_GIST = /^[a-f0-9]{20,64}$/;
|
|
45
|
+
const OWNER_RE = /^[A-Za-z0-9-]{1,39}$/;
|
|
11
46
|
const compareCanonicalText = (left, right) => String(left) < String(right) ? -1
|
|
12
47
|
: String(left) > String(right) ? 1 : 0;
|
|
13
48
|
|
|
14
|
-
const withoutReceiptDigest = ({ receiptSha256: _receiptSha256, ...payload }) => payload;
|
|
15
|
-
|
|
16
49
|
function validDate(value) {
|
|
17
50
|
return typeof value === 'string' && Number.isFinite(Date.parse(value));
|
|
18
51
|
}
|
|
19
52
|
|
|
53
|
+
function validateFilename(filename) {
|
|
54
|
+
return typeof filename === 'string' && Boolean(filename) && !filename.includes('\0')
|
|
55
|
+
&& !filename.split(/[\\/]/).some((segment) => !segment || segment === '.' || segment === '..');
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// ── per-gist receipt sealing (schema 3) ──────────────────────────────────────────────────────────
|
|
20
59
|
export function sealGistReceipt(receipt) {
|
|
21
60
|
const files = [...(receipt.files || [])].sort((a, b) => compareCanonicalText(a.filename, b.filename));
|
|
22
61
|
const payload = {
|
|
@@ -34,6 +73,8 @@ export function sealGistReceipt(receipt) {
|
|
|
34
73
|
return { ...payload, receiptSha256: digest(payload) };
|
|
35
74
|
}
|
|
36
75
|
|
|
76
|
+
// ── aggregate receipt sealing (schema 3). The ONLY sealing point: passagesSha256 is always bound
|
|
77
|
+
// here from the actually-written passage bytes — never sealed null, never bound in a second step.
|
|
37
78
|
export function sealGistReceiptSet(receipt) {
|
|
38
79
|
const ids = Object.keys(receipt.gists || {}).sort();
|
|
39
80
|
const gists = Object.fromEntries(ids.map((id) => [id, receipt.gists[id]]));
|
|
@@ -57,17 +98,155 @@ export function sealGistReceiptSet(receipt) {
|
|
|
57
98
|
return { ...payload, receiptSha256: digest(payload) };
|
|
58
99
|
}
|
|
59
100
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
101
|
+
// ── transport: per-gist detail + rendering primitives ────────────────────────────────────────────
|
|
102
|
+
//
|
|
103
|
+
// Actions' default GITHUB_TOKEN is a GitHub App token with no gist scope ("Resource not accessible
|
|
104
|
+
// by integration", HTTP 403). corpus-seed.yml exports GH_TOKEN from a narrowly-scoped
|
|
105
|
+
// RUVNET_GISTS_TOKEN secret for the reconcile step, but a nightly/ad-hoc invocation may still hit an
|
|
106
|
+
// unscoped token. `gh` itself already reads GH_TOKEN/GITHUB_TOKEN from the environment (verified
|
|
107
|
+
// live: `gh help environment`, 2026-09-13), so no code here chooses a token.
|
|
108
|
+
//
|
|
109
|
+
// Failures observed against the real API fall into four shapes, each handled differently:
|
|
110
|
+
// - a moved/deleted gist (404) -> thrown immediately as a typed, non-retryable
|
|
111
|
+
// GistFetchError. Never returned as null.
|
|
112
|
+
// - "resource not accessible by integration" -> thrown non-retryably BY defaultFetchGist, then
|
|
113
|
+
// (still-missing gist scope, 403) caught ONE layer up by defaultFetchDetail and
|
|
114
|
+
// retried against the PUBLIC, unauthenticated
|
|
115
|
+
// detail endpoint (no Authorization header) —
|
|
116
|
+
// mirroring source-coverage.mjs's
|
|
117
|
+
// listGistsUnauthenticated fallback pattern.
|
|
118
|
+
// - a rate limit (primary or secondary, 403) -> retried with backoff, bounded by `retries`.
|
|
119
|
+
// - any other transient failure (timeouts, -> retried with backoff, bounded by `retries`.
|
|
120
|
+
// TLS/connection resets, 5xx, 429)
|
|
121
|
+
const RATE_LIMIT_RE = /rate limit/i;
|
|
122
|
+
const NOT_FOUND_RE = /\bnot found\b|HTTP 404/i;
|
|
123
|
+
const FORBIDDEN_INTEGRATION_RE = /resource not accessible by integration/i;
|
|
124
|
+
const TRANSIENT_RE = /timeout|timed out|TLS handshake|ECONNRESET|ECONNREFUSED|EAI_AGAIN|temporary failure|HTTP 5\d\d|HTTP 429|socket hang up/i;
|
|
125
|
+
|
|
126
|
+
export class GistFetchError extends Error {
|
|
127
|
+
constructor(message, { code, gistId, status, retryable = false } = {}) {
|
|
128
|
+
super(message);
|
|
129
|
+
this.name = 'GistFetchError';
|
|
130
|
+
this.code = code;
|
|
131
|
+
this.gistId = gistId;
|
|
132
|
+
if (status !== undefined) this.status = status;
|
|
133
|
+
this.retryable = retryable;
|
|
134
|
+
}
|
|
64
135
|
}
|
|
65
136
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
137
|
+
function classifyGistFetchFailure(stderr, gistId) {
|
|
138
|
+
const message = String(stderr || '').trim();
|
|
139
|
+
if (NOT_FOUND_RE.test(message)) {
|
|
140
|
+
return new GistFetchError(`gist ${gistId} was moved or deleted: ${message}`,
|
|
141
|
+
{ code: 'GIST_NOT_FOUND', gistId, retryable: false });
|
|
142
|
+
}
|
|
143
|
+
if (FORBIDDEN_INTEGRATION_RE.test(message)) {
|
|
144
|
+
return new GistFetchError(`gist ${gistId} fetch rejected -- token lacks gist scope: ${message}`,
|
|
145
|
+
{ code: 'GIST_FORBIDDEN', gistId, retryable: false });
|
|
146
|
+
}
|
|
147
|
+
if (RATE_LIMIT_RE.test(message)) {
|
|
148
|
+
return new GistFetchError(`gist ${gistId} fetch rate-limited: ${message}`,
|
|
149
|
+
{ code: 'GIST_RATE_LIMITED', gistId, retryable: true });
|
|
150
|
+
}
|
|
151
|
+
if (TRANSIENT_RE.test(message)) {
|
|
152
|
+
return new GistFetchError(`gist ${gistId} fetch failed transiently: ${message}`,
|
|
153
|
+
{ code: 'GIST_TRANSIENT_FAILURE', gistId, retryable: true });
|
|
154
|
+
}
|
|
155
|
+
return new GistFetchError(`gist ${gistId} fetch failed: ${message}`,
|
|
156
|
+
{ code: 'GIST_FETCH_FAILED', gistId, retryable: false });
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function defaultSleep(ms) {
|
|
160
|
+
return new Promise((resolve) => { setTimeout(resolve, ms); });
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function abortError(signal) {
|
|
164
|
+
return signal?.reason instanceof Error ? signal.reason : new DOMException('gist fetch aborted', 'AbortError');
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// UNCHANGED from Step 0 -- retry/error-classification logic here is already correct and tested.
|
|
168
|
+
// `defaultFetchDetail` (below) wraps this rather than duplicating or modifying its behavior.
|
|
169
|
+
export async function defaultFetchGist(id, { spawn = spawnSync, retries = 3, retryDelayMs = 300,
|
|
170
|
+
sleep = defaultSleep, signal } = {}) {
|
|
171
|
+
if (signal?.aborted) throw abortError(signal);
|
|
172
|
+
const attempts = Math.max(1, retries);
|
|
173
|
+
let lastError;
|
|
174
|
+
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
175
|
+
const result = spawn('gh', ['api', `gists/${id}`], { encoding: 'utf8', maxBuffer: 64 * 1024 * 1024 });
|
|
176
|
+
if (result.status === 0) return JSON.parse(result.stdout);
|
|
177
|
+
lastError = classifyGistFetchFailure(result.stderr, id);
|
|
178
|
+
if (!lastError.retryable || attempt === attempts) throw lastError;
|
|
179
|
+
await sleep(retryDelayMs * attempt);
|
|
180
|
+
if (signal?.aborted) throw abortError(signal);
|
|
181
|
+
}
|
|
182
|
+
throw lastError;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// The unauthenticated fallback used ONLY for the specific 403 "no gist scope" rejection. Mirrors
|
|
186
|
+
// source-coverage.mjs's listGistsUnauthenticated: same env var, same accept header, same
|
|
187
|
+
// unauthenticated-quota reasoning. `fetchImpl` is a test seam (default: global fetch).
|
|
188
|
+
async function fetchPublicGistDetail(id, { fetchImpl = globalThis.fetch, signal } = {}) {
|
|
189
|
+
const apiBase = process.env.RUVNET_GISTS_API || 'https://api.github.com';
|
|
190
|
+
const response = await fetchImpl(`${apiBase}/gists/${id}`, {
|
|
191
|
+
headers: { accept: 'application/vnd.github+json', 'user-agent': 'ruvnet-brain-gist-receipts' },
|
|
192
|
+
signal,
|
|
193
|
+
});
|
|
194
|
+
if (!response.ok) {
|
|
195
|
+
throw new GistFetchError(`gist ${id} public detail fetch failed: HTTP ${response.status}`,
|
|
196
|
+
{ code: 'GIST_FETCH_FAILED', gistId: id, status: response.status, retryable: false });
|
|
197
|
+
}
|
|
198
|
+
return response.json();
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// The transport `captureGistSources` uses by default for per-gist DETAIL. Wraps `defaultFetchGist`
|
|
202
|
+
// without duplicating its retry logic: on any OTHER classified failure (not found, rate-limited,
|
|
203
|
+
// transient, or a plain non-retryable failure) the original error propagates untouched -- only the
|
|
204
|
+
// specific integration-token rejection falls back to the public endpoint.
|
|
205
|
+
export async function defaultFetchDetail(id, options = {}) {
|
|
206
|
+
try {
|
|
207
|
+
return await defaultFetchGist(id, options);
|
|
208
|
+
} catch (error) {
|
|
209
|
+
if (error?.code !== 'GIST_FORBIDDEN') throw error;
|
|
210
|
+
return fetchPublicGistDetail(id, options);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const RAW_TRANSIENT_RE = /timeout|timed out|ECONNRESET|ECONNREFUSED|EAI_AGAIN|socket hang up|HTTP 5\d\d|HTTP 429/i;
|
|
215
|
+
|
|
216
|
+
// RAW-CONTENT transport, kept on its own retry/backoff budget from `defaultFetchDetail` above --
|
|
217
|
+
// per-gist API detail requests and raw-content requests are different quotas on GitHub's side and
|
|
218
|
+
// must never share one guessed retry count. Non-truncated files never hit the network at all (the
|
|
219
|
+
// detail response already carries their content).
|
|
220
|
+
export async function defaultFetchRaw(file, { fetchImpl = globalThis.fetch, retries = 3,
|
|
221
|
+
retryDelayMs = 300, sleep = defaultSleep, signal } = {}) {
|
|
222
|
+
if (!file?.truncated) return Buffer.from(file?.content || '', 'utf8');
|
|
223
|
+
if (signal?.aborted) throw abortError(signal);
|
|
224
|
+
const attempts = Math.max(1, retries);
|
|
225
|
+
let lastError;
|
|
226
|
+
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
227
|
+
try {
|
|
228
|
+
const response = await fetchImpl(file.raw_url, {
|
|
229
|
+
headers: { 'user-agent': 'ruvnet-brain-gist-receipts' }, signal,
|
|
230
|
+
});
|
|
231
|
+
if (response.ok) return Buffer.from(await response.arrayBuffer());
|
|
232
|
+
const transient = response.status === 429 || (response.status >= 500 && response.status < 600);
|
|
233
|
+
lastError = new GistFetchError(`raw gist fetch failed: HTTP ${response.status}`,
|
|
234
|
+
{ code: transient ? 'GIST_RAW_TRANSIENT_FAILURE' : 'GIST_RAW_FETCH_FAILED', status: response.status, retryable: transient });
|
|
235
|
+
if (!transient) throw lastError;
|
|
236
|
+
} catch (error) {
|
|
237
|
+
if (error instanceof GistFetchError) { lastError = error; if (!error.retryable) throw error; }
|
|
238
|
+
else {
|
|
239
|
+
const transient = RAW_TRANSIENT_RE.test(String(error?.message || ''));
|
|
240
|
+
lastError = new GistFetchError(`raw gist fetch failed: ${error.message}`,
|
|
241
|
+
{ code: transient ? 'GIST_RAW_TRANSIENT_FAILURE' : 'GIST_RAW_FETCH_FAILED', retryable: transient });
|
|
242
|
+
if (!transient) throw lastError;
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
if (attempt === attempts) throw lastError;
|
|
246
|
+
await sleep(retryDelayMs * attempt);
|
|
247
|
+
if (signal?.aborted) throw abortError(signal);
|
|
248
|
+
}
|
|
249
|
+
throw lastError;
|
|
71
250
|
}
|
|
72
251
|
|
|
73
252
|
function listedFileIdentity(gist) {
|
|
@@ -77,83 +256,111 @@ function listedFileIdentity(gist) {
|
|
|
77
256
|
}));
|
|
78
257
|
}
|
|
79
258
|
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
if (
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
return
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
const filenames = row.files.map(({ filename }) => filename);
|
|
100
|
-
return new Set(filenames).size === filenames.length
|
|
101
|
-
&& canonicalJson(filenames) === canonicalJson([...filenames].sort(compareCanonicalText))
|
|
102
|
-
&& canonicalJson(row) === canonicalJson(sealGistReceipt(row));
|
|
103
|
-
}
|
|
104
|
-
|
|
105
|
-
export function validateGistReceiptSet(receipt, observation) {
|
|
106
|
-
const ids = observation?.gists?.rows?.map(({ id }) => String(id)).sort() || [];
|
|
107
|
-
const received = Object.keys(receipt?.gists || {}).sort();
|
|
108
|
-
if (receipt?.schemaVersion !== 3 || receipt?.kind !== 'ruvnet-brain-gist-source-receipts'
|
|
109
|
-
|| receipt.owner !== observation?.owner || !validDate(receipt.generated)
|
|
110
|
-
|| !validDate(receipt.observedAt)
|
|
111
|
-
|| canonicalJson(ids) !== canonicalJson(received)
|
|
112
|
-
|| receipt.gistSet?.count !== ids.length || receipt.gistSet?.idsSha256 !== digest(ids)
|
|
113
|
-
|| receipt.gistSet?.observationSha256 !== observation?.observationSha256
|
|
114
|
-
|| receipt.sourceObservationSha256 !== observation?.observationSha256) {
|
|
115
|
-
throw new Error('gist receipt set does not exactly match the sealed source observation');
|
|
116
|
-
}
|
|
117
|
-
for (const id of ids) {
|
|
118
|
-
const row = receipt.gists[id];
|
|
119
|
-
const stub = observation.gists.rows.find((gist) => String(gist.id) === id);
|
|
120
|
-
if (row?.schemaVersion !== 3 || row?.kind !== 'ruvnet-brain-gist-source-receipt'
|
|
121
|
-
|| row?.gistId !== id || row?.complete !== true || row.updatedAt !== stub?.updated_at
|
|
122
|
-
|| !HEX40.test(String(row.versionSha || ''))
|
|
123
|
-
|| !validDate(row.updatedAt) || !validDate(row.ingestedAt)
|
|
124
|
-
|| !Array.isArray(row.files) || row.fileCount !== row.files.length || row.files.length === 0
|
|
125
|
-
|| row.files.some((file) => !exactFileIdentity(file))) throw new Error(`gist ${id} receipt is incomplete`);
|
|
126
|
-
const filenames = row.files.map(({ filename }) => filename);
|
|
127
|
-
if (filenames.some((filename) => typeof filename !== 'string' || !filename)
|
|
128
|
-
|| new Set(filenames).size !== filenames.length
|
|
129
|
-
|| canonicalJson(filenames) !== canonicalJson([...filenames].sort(compareCanonicalText))) {
|
|
130
|
-
throw new Error(`gist ${id} file inventory is not unique and sorted`);
|
|
259
|
+
// ── rendering primitives -- the ONE banner/chunk/JSONL implementation. Moved here from
|
|
260
|
+
// rebuild-gists-from-receipts.mjs, which now imports (never reimplements) these.
|
|
261
|
+
export function rawUrlFor({ owner, gistId, versionSha, filename }) {
|
|
262
|
+
if (!OWNER_RE.test(String(owner || '')) || !HEX_GIST.test(String(gistId || ''))
|
|
263
|
+
|| !HEX40.test(String(versionSha || ''))) throw new Error('cannot construct raw URL from malformed source identity');
|
|
264
|
+
if (!validateFilename(filename)) throw new Error(`gist ${gistId} has an unsafe or missing filename`);
|
|
265
|
+
const encodedFile = filename.split('/').map(encodeURIComponent).join('/');
|
|
266
|
+
return `https://gist.githubusercontent.com/${owner}/${gistId}/raw/${versionSha}/${encodedFile}`;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
export function paragraphChunks(text, size = 3200) {
|
|
270
|
+
if (typeof text !== 'string') throw new Error('passage content must be text');
|
|
271
|
+
if (!Number.isSafeInteger(size) || size < 1) throw new Error('chunk size must be a positive integer');
|
|
272
|
+
const out = [];
|
|
273
|
+
let buffer = '';
|
|
274
|
+
for (const paragraph of text.split(/\n\n+/)) {
|
|
275
|
+
if (buffer && buffer.length + paragraph.length + 2 > size) {
|
|
276
|
+
out.push(buffer);
|
|
277
|
+
buffer = '';
|
|
131
278
|
}
|
|
132
|
-
|
|
133
|
-
if (row.receiptSha256 !== digest(withoutReceiptDigest(row))) throw new Error(`gist ${id} receipt digest differs`);
|
|
279
|
+
buffer = buffer ? `${buffer}\n\n${paragraph}` : paragraph;
|
|
134
280
|
}
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
281
|
+
if (buffer.trim()) out.push(buffer);
|
|
282
|
+
return out;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
export function provenanceBanner({ owner, gistId, filename, updatedAt }) {
|
|
286
|
+
return `SOURCE: GitHub gist by @${owner} — "${filename.replace(/\s+/g, ' ').trim().slice(0, 160)}"\n`
|
|
287
|
+
+ `GIST STATUS: rUv's own notes / release announcement — may describe PROPOSED or UNRELEASED work.\n`
|
|
288
|
+
+ `Treat as intent, not as confirmed shipped behavior: verify against repo source before asserting.\n`
|
|
289
|
+
+ `updated: ${updatedAt.slice(0, 10)} · https://gist.github.com/${owner}/${gistId}\n\n`;
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
const jsonLine = (value) => JSON.stringify(value).replace(/\u2028/g, '\\u2028').replace(/\u2029/g, '\\u2029');
|
|
293
|
+
|
|
294
|
+
// ── stage 1: capture ─────────────────────────────────────────────────────────────────────────────
|
|
295
|
+
//
|
|
296
|
+
// `cache`, when supplied, is a PRIOR CapturedGistSet (this function's own return shape, including
|
|
297
|
+
// raw body text) -- never the published, body-free receipt (a receipt alone can never supply
|
|
298
|
+
// missing body bytes, which is exactly the bug this whole consolidation fixes). A cached gist is
|
|
299
|
+
// reused ONLY when its identity (id/updatedAt/complete) matches AND every included file's body
|
|
300
|
+
// re-hashes to its recorded sha256/bytes. Identity that legitimately changed (a different
|
|
301
|
+
// updatedAt) is an ordinary cache MISS and triggers a fresh fetch; identity that CLAIMS to be
|
|
302
|
+
// current but whose body hash does not verify is treated as a CORRUPTED/TAMPERED cache entry and
|
|
303
|
+
// throws rather than silently trusting or silently refetching it.
|
|
304
|
+
function reusableCachedGist(cached, stub) {
|
|
305
|
+
if (!cached || cached.gistId !== stub.id || cached.updatedAt !== stub.updated_at
|
|
306
|
+
|| cached.complete !== true || !Array.isArray(cached.files) || !cached.files.length) {
|
|
307
|
+
return 'miss';
|
|
139
308
|
}
|
|
140
|
-
|
|
141
|
-
|
|
309
|
+
// A matching gistId/updatedAt/complete and verified body hashes for the files the cache HAPPENS TO
|
|
310
|
+
// CARRY prove nothing about whether the gist's file set itself has changed since capture -- the
|
|
311
|
+
// cache could be missing a file that was added, or still carrying one that was removed, with
|
|
312
|
+
// updated_at untouched in between (Dual's 2026-09-13 pass proved this exact gap: a stale cache with
|
|
313
|
+
// a matching updatedAt and valid, correctly-hashed bodies for its OWN recorded files still returned
|
|
314
|
+
// 'reuse' even though the stub's current file listing had already diverged). `stub.files` (the
|
|
315
|
+
// current list-observation) is compared against the cache's own recorded filenames -- the one piece
|
|
316
|
+
// of per-file identity the cache actually stores for every file, included or excluded alike (the
|
|
317
|
+
// richer per-file identity `listedFileIdentity` compares -- rawUrl/size/type/language -- is captured
|
|
318
|
+
// fresh from `full` at fetch time via the GIST_OBSERVATION_MOVED check below and is NEVER persisted
|
|
319
|
+
// into the cache record itself, so it is not available here without a live fetch; a name-identical
|
|
320
|
+
// but otherwise-altered file is still caught downstream by that same check once this cache entry is
|
|
321
|
+
// correctly treated as a miss and refetched). A mismatch here is an ordinary MISS -- the gist
|
|
322
|
+
// genuinely changed -- never 'tampered', which is reserved for a body-hash failure on content the
|
|
323
|
+
// cache claims is still current.
|
|
324
|
+
const cachedFilenames = cached.files.map((file) => file.filename).sort(compareCanonicalText);
|
|
325
|
+
const stubFilenames = Object.keys(stub?.files || {}).sort(compareCanonicalText);
|
|
326
|
+
if (cachedFilenames.length !== stubFilenames.length
|
|
327
|
+
|| cachedFilenames.some((name, index) => name !== stubFilenames[index])) {
|
|
328
|
+
return 'miss';
|
|
329
|
+
}
|
|
330
|
+
// A body-free cache entry (e.g. a caller naively handing in the PUBLISHED, body-free receipt) can
|
|
331
|
+
// never supply reuse -- "a receipt or timestamp alone can never supply missing body bytes" is the
|
|
332
|
+
// whole reason capture-cache reuse now requires real bytes. That is an ordinary miss, not tampering:
|
|
333
|
+
// tampering is specifically a cache that CLAIMS to carry a body whose hash does not verify.
|
|
334
|
+
if (cached.files.some((file) => file.included === true && typeof file.body !== 'string')) return 'miss';
|
|
335
|
+
for (const file of cached.files) {
|
|
336
|
+
if (file.included !== true) continue;
|
|
337
|
+
const bodyBuffer = Buffer.from(file.body, 'utf8');
|
|
338
|
+
if (bodyBuffer.length !== file.bytes || sha256(bodyBuffer) !== file.sha256) return 'tampered';
|
|
339
|
+
}
|
|
340
|
+
return 'reuse';
|
|
142
341
|
}
|
|
143
342
|
|
|
144
|
-
export async function
|
|
145
|
-
|
|
343
|
+
export async function captureGistSources({ observation, cache = null, fetchDetail = defaultFetchDetail,
|
|
344
|
+
fetchRaw = defaultFetchRaw, signal, now = () => new Date().toISOString() } = {}) {
|
|
146
345
|
const stubs = observation?.gists?.rows;
|
|
147
|
-
if (!Array.isArray(stubs) || stubs.some(({ id }) =>
|
|
346
|
+
if (!Array.isArray(stubs) || stubs.some(({ id }) => !HEX_GIST.test(String(id || '')))
|
|
148
347
|
|| new Set(stubs.map(({ id }) => id)).size !== stubs.length) {
|
|
149
348
|
throw new Error('source observation has missing or duplicate gist ids');
|
|
150
349
|
}
|
|
151
350
|
const gists = {};
|
|
351
|
+
const reused = [];
|
|
352
|
+
const fetched = [];
|
|
152
353
|
for (const stub of [...stubs].sort((a, b) => String(a.id).localeCompare(String(b.id)))) {
|
|
153
|
-
|
|
154
|
-
const
|
|
155
|
-
|
|
156
|
-
|
|
354
|
+
if (signal?.aborted) throw abortError(signal);
|
|
355
|
+
const cached = cache?.gists?.[stub.id];
|
|
356
|
+
const verdict = cached ? reusableCachedGist(cached, stub) : 'miss';
|
|
357
|
+
if (verdict === 'tampered') {
|
|
358
|
+
throw new Error(`capture cache for gist ${stub.id} failed body-hash verification -- refusing `
|
|
359
|
+
+ 'to reuse a tampered or corrupted cache entry');
|
|
360
|
+
}
|
|
361
|
+
if (verdict === 'reuse') { gists[stub.id] = cached; reused.push(stub.id); continue; }
|
|
362
|
+
|
|
363
|
+
const full = await fetchDetail(stub.id, { signal });
|
|
157
364
|
const versionSha = full?.history?.[0]?.version;
|
|
158
365
|
if (full?.id !== stub.id || full.updated_at !== stub.updated_at || !HEX40.test(String(versionSha || ''))
|
|
159
366
|
|| canonicalJson(listedFileIdentity(full)) !== canonicalJson(listedFileIdentity(stub))) {
|
|
@@ -168,16 +375,204 @@ export async function reconcileGistReceipts({ observation, existing = null, fetc
|
|
|
168
375
|
files.push({ filename, included: false, reason: 'non-text policy exclusion', size: file.size ?? null });
|
|
169
376
|
continue;
|
|
170
377
|
}
|
|
171
|
-
const
|
|
172
|
-
const body = Buffer.isBuffer(
|
|
173
|
-
|
|
378
|
+
const rawBody = await fetchRaw(file, { signal });
|
|
379
|
+
const body = Buffer.isBuffer(rawBody) ? rawBody : Buffer.from(String(rawBody), 'utf8');
|
|
380
|
+
let text;
|
|
381
|
+
try { text = new TextDecoder('utf-8', { fatal: true }).decode(body); }
|
|
174
382
|
catch { throw new Error(`gist ${stub.id}/${filename} is not valid UTF-8 text`); }
|
|
175
|
-
files.push({ filename, included: true, sha256: sha256(body), bytes: body.length });
|
|
383
|
+
files.push({ filename, included: true, sha256: sha256(body), bytes: body.length, body: text });
|
|
176
384
|
}
|
|
177
|
-
gists[stub.id] =
|
|
178
|
-
|
|
385
|
+
gists[stub.id] = { gistId: stub.id, versionSha, updatedAt: full.updated_at, ingestedAt: now(),
|
|
386
|
+
complete: files.length === Object.keys(full.files || {}).length, files };
|
|
387
|
+
fetched.push(stub.id);
|
|
388
|
+
}
|
|
389
|
+
return {
|
|
390
|
+
owner: observation.owner,
|
|
391
|
+
observedAt: observation.observedAt,
|
|
392
|
+
sourceObservationSha256: observation.observationSha256,
|
|
393
|
+
generatedAt: now(),
|
|
394
|
+
gists,
|
|
395
|
+
reuseEvidence: { reused, fetched },
|
|
396
|
+
};
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
// ── stage 2: render ──────────────────────────────────────────────────────────────────────────────
|
|
400
|
+
export function renderGistPassages({ captured, generatedAt } = {}) {
|
|
401
|
+
if (!captured?.gists || typeof captured.gists !== 'object') throw new Error('renderGistPassages requires a captured gist set');
|
|
402
|
+
const owner = captured.owner;
|
|
403
|
+
const passages = [];
|
|
404
|
+
const entries = {};
|
|
405
|
+
const gistRecords = {};
|
|
406
|
+
let id = 0;
|
|
407
|
+
for (const gistId of Object.keys(captured.gists).sort()) {
|
|
408
|
+
const gist = captured.gists[gistId];
|
|
409
|
+
for (const file of [...(gist.files || [])].sort((a, b) => compareCanonicalText(a.filename, b.filename))) {
|
|
410
|
+
if (!file.included) continue;
|
|
411
|
+
const chunks = paragraphChunks(file.body);
|
|
412
|
+
const title = file.filename.replace(/\s+/g, ' ').trim().slice(0, 180) || file.filename;
|
|
413
|
+
const banner = provenanceBanner({ owner, gistId, filename: file.filename, updatedAt: gist.updatedAt });
|
|
414
|
+
for (const [chunkIndex, chunk] of chunks.entries()) {
|
|
415
|
+
const passageId = String(id++);
|
|
416
|
+
const passagePath = `${gistId.slice(0, 8)}/${file.filename}${chunks.length > 1 ? `#${chunkIndex}` : ''}`;
|
|
417
|
+
const text = banner + chunk;
|
|
418
|
+
passages.push({ id: passageId, text, path: passagePath, title });
|
|
419
|
+
entries[passageId] = { path: passagePath, kind: 'doc', title, chunk: chunkIndex, preview: text.slice(0, 200) };
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
// The publishable per-gist receipt row never carries raw body text -- strip it here, once, at
|
|
423
|
+
// the render boundary, rather than trusting every caller to remember to drop it.
|
|
424
|
+
gistRecords[gistId] = sealGistReceipt({
|
|
425
|
+
...gist, files: (gist.files || []).map(({ body: _body, ...rest }) => rest),
|
|
426
|
+
});
|
|
427
|
+
}
|
|
428
|
+
const passageBytes = Buffer.from(`${passages.map(jsonLine).join('\n')}\n`, 'utf8');
|
|
429
|
+
const metadata = {
|
|
430
|
+
model: 'ruv-gists', dimensions: 0, metric: 'cosine', name: 'ruv-gists', generated: generatedAt,
|
|
431
|
+
repo: `gists/${owner}`,
|
|
432
|
+
note: "rUv's public gists — announcements and thinking, PROPOSED unless confirmed in repo source.",
|
|
433
|
+
entries,
|
|
434
|
+
};
|
|
435
|
+
return { passageBytes, metadata, gistRecords };
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
// ── the one shared validator: capture-cache reuse, producer output, and archive verification all
|
|
439
|
+
// go through this. Wraps (never reimplements) coverage-integrity.mjs's validateGistAggregateReceipt
|
|
440
|
+
// for schema/digest/passage-binding, and adds the checks that validator does not already cover:
|
|
441
|
+
// owner shape, date validity, and safe (non-path-traversal) filenames.
|
|
442
|
+
//
|
|
443
|
+
// `mode` controls only how strictly the receipt is tied to a live `observation`:
|
|
444
|
+
// 'produce' (default) — a freshly built receipt MUST match the exact supplied observation
|
|
445
|
+
// (gist id set + sourceObservationSha256).
|
|
446
|
+
// 'reuse' — an existing on-disk receipt is being checked as a candidate for capture
|
|
447
|
+
// reuse against the CURRENT observation; same strictness as 'produce'.
|
|
448
|
+
// 'archive' — a shipped/published receipt is being independently verified and no live
|
|
449
|
+
// observation is available; id-set/observation binding is not enforced.
|
|
450
|
+
// In every mode the receipt must bind its ACTUAL on-disk passage bytes (`passagesFile`) — that
|
|
451
|
+
// binding is never optional, which is the direct fix for the passagesSha256:null bug.
|
|
452
|
+
export function validateGistReceipt({ receipt, observation = null, passagesFile,
|
|
453
|
+
expectedOwner = null, mode = 'produce' } = {}) {
|
|
454
|
+
if (!passagesFile) throw new Error('gist receipt validation requires the receipt\'s passages file');
|
|
455
|
+
const owner = expectedOwner ?? observation?.owner ?? receipt?.owner;
|
|
456
|
+
if (!OWNER_RE.test(String(owner || ''))) throw new Error('gist receipt owner is missing or malformed');
|
|
457
|
+
if (receipt?.owner !== owner) throw new Error('gist receipt owner does not match the expected owner');
|
|
458
|
+
if (!validDate(receipt?.generated) || !validDate(receipt?.observedAt)) {
|
|
459
|
+
throw new Error('gist receipt has an invalid generated/observedAt timestamp');
|
|
460
|
+
}
|
|
461
|
+
for (const [gistId, row] of Object.entries(receipt?.gists || {})) {
|
|
462
|
+
if (!validDate(row?.updatedAt) || !validDate(row?.ingestedAt)) {
|
|
463
|
+
throw new Error(`gist ${gistId} has an invalid updatedAt/ingestedAt timestamp`);
|
|
464
|
+
}
|
|
465
|
+
for (const file of row?.files || []) {
|
|
466
|
+
if (!validateFilename(file?.filename)) throw new Error(`gist ${gistId} has an unsafe or missing filename`);
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
const bindObservation = mode !== 'archive';
|
|
470
|
+
const expectedIds = bindObservation && observation?.gists?.rows
|
|
471
|
+
? observation.gists.rows.map(({ id }) => String(id)) : null;
|
|
472
|
+
const sourceObservationSha256 = bindObservation ? (observation?.observationSha256 ?? null) : null;
|
|
473
|
+
validateGistAggregateReceipt({ receipt, passagesFile, expectedIds, sourceObservationSha256 });
|
|
474
|
+
return receipt;
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
// ── stage 3: build ───────────────────────────────────────────────────────────────────────────────
|
|
478
|
+
function defaultBuildVector({ root, assetsDir, store }) {
|
|
479
|
+
const script = path.join(path.resolve(root), 'kb', 'forge-big.mjs');
|
|
480
|
+
const result = spawnSync(process.execPath, [script, 'both', '--dir', assetsDir, '--name', store], {
|
|
481
|
+
encoding: 'utf8', stdio: 'inherit', env: { ...process.env },
|
|
482
|
+
});
|
|
483
|
+
if (result.error || result.status !== 0) {
|
|
484
|
+
throw new Error(`${store} vector build failed (${result.error?.message || `exit ${result.status}`})`);
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
function writeFileAtomic(file, content) {
|
|
489
|
+
const temporary = `${file}.tmp-${process.pid}-${crypto.randomBytes(6).toString('hex')}`;
|
|
490
|
+
fs.writeFileSync(temporary, content, { flag: 'wx' });
|
|
491
|
+
fs.renameSync(temporary, file);
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
/**
|
|
495
|
+
* buildGistAggregate — capture + render + write + (optionally) embed + seal + validate, atomically.
|
|
496
|
+
*
|
|
497
|
+
* `outDir` is the LIVE directory the result is published into. All work happens in a FRESH stage
|
|
498
|
+
* directory first; nothing is written into `outDir` until the complete result (passages, receipt,
|
|
499
|
+
* and — when `buildVector` is supplied — the RVF family) validates. A failed vector build therefore
|
|
500
|
+
* never exposes a finalized receipt: the stage directory is discarded and `outDir` is untouched.
|
|
501
|
+
*
|
|
502
|
+
* `buildVector`: defaults to the real forge-big.mjs embed. Pass `null` to skip embedding entirely
|
|
503
|
+
* (e.g. a cheap nightly content refresh that defers embedding to a separate sharded job) — in that
|
|
504
|
+
* case `generation` on the returned StoreResult is null and RVF-GENERATIONS.json is left untouched.
|
|
505
|
+
*
|
|
506
|
+
* An empty observed gist set (owner currently has zero gists) OMITS the aggregate entirely: no
|
|
507
|
+
* ruv-gists.sources.json, no ruv-gists store, nothing written or promoted. A nonempty observed set
|
|
508
|
+
* that yields zero usable passages after non-text exclusions FAILS EXPLICITLY -- no known,
|
|
509
|
+
* independently-defined policy in this repo permits shipping a zero-passage nonempty aggregate.
|
|
510
|
+
*/
|
|
511
|
+
export async function buildGistAggregate({ observation, cache = null, outDir, root = DEFAULT_ROOT, transport = {},
|
|
512
|
+
buildVector = defaultBuildVector, sourceCommit = null, signal, now = () => new Date().toISOString() } = {}) {
|
|
513
|
+
const live = path.resolve(outDir || '');
|
|
514
|
+
const stubs = observation?.gists?.rows || [];
|
|
515
|
+
if (!stubs.length) {
|
|
516
|
+
return { store: 'ruv-gists', kind: 'gist-aggregate', omitted: true, generation: null,
|
|
517
|
+
sourceMetadata: null, files: [], sourceReceipt: null, reuseEvidence: { reused: [], fetched: [] } };
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
const captured = await captureGistSources({
|
|
521
|
+
observation, cache, fetchDetail: transport.fetchDetail, fetchRaw: transport.fetchRaw, signal, now,
|
|
522
|
+
});
|
|
523
|
+
const { passageBytes, metadata, gistRecords } = renderGistPassages({ captured, generatedAt: captured.generatedAt });
|
|
524
|
+
const includedFileCount = Object.values(captured.gists)
|
|
525
|
+
.reduce((total, gist) => total + gist.files.filter((file) => file.included).length, 0);
|
|
526
|
+
if (includedFileCount === 0) {
|
|
527
|
+
throw new Error('ruv-gists aggregate: observed gist set is nonempty but produced zero usable '
|
|
528
|
+
+ 'passages after non-text exclusions -- refusing to seal a degenerate aggregate');
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
fs.mkdirSync(path.dirname(live), { recursive: true });
|
|
532
|
+
const stage = fs.mkdtempSync(path.join(path.dirname(live), '.gist-aggregate-'));
|
|
533
|
+
try {
|
|
534
|
+
const passagesFile = path.join(stage, 'ruv-gists.passages.jsonl');
|
|
535
|
+
const metaFile = path.join(stage, 'ruv-gists.meta.json');
|
|
536
|
+
const sourcesFile = path.join(stage, 'ruv-gists.sources.json');
|
|
537
|
+
writeFileAtomic(passagesFile, passageBytes);
|
|
538
|
+
writeFileAtomic(metaFile, `${JSON.stringify(metadata, null, 2)}\n`);
|
|
539
|
+
|
|
540
|
+
// Hash the ACTUALLY WRITTEN bytes on disk -- never the in-memory buffer or a claimed value.
|
|
541
|
+
const passagesSha256 = crypto.createHash('sha256').update(fs.readFileSync(passagesFile)).digest('hex');
|
|
542
|
+
const sourceReceipt = sealGistReceiptSet({
|
|
543
|
+
owner: captured.owner, generated: captured.generatedAt, observedAt: captured.observedAt,
|
|
544
|
+
sourceObservationSha256: captured.sourceObservationSha256, passagesSha256, gists: gistRecords,
|
|
545
|
+
});
|
|
546
|
+
writeFileAtomic(sourcesFile, `${JSON.stringify(sourceReceipt, null, 2)}\n`);
|
|
547
|
+
|
|
548
|
+
let generation = null;
|
|
549
|
+
const promoted = ['ruv-gists.passages.jsonl', 'ruv-gists.meta.json', 'ruv-gists.sources.json'];
|
|
550
|
+
if (buildVector) {
|
|
551
|
+
await buildVector({ root, assetsDir: stage, store: 'ruv-gists' });
|
|
552
|
+
generation = writeRvfGeneration({
|
|
553
|
+
dir: stage, previousDir: fs.existsSync(live) ? live : stage, store: 'ruv-gists',
|
|
554
|
+
model: EMBED_MODEL, dimensions: EMBED_DIMENSIONS, sourceCommit, builtUtc: now(),
|
|
555
|
+
});
|
|
556
|
+
promoted.push('ruv-gists.big.rvf', 'ruv-gists.big.rvf.idmap.json', 'ruv-gists.big.rvf.embed.json', 'RVF-GENERATIONS.json');
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
// Validate the complete, staged result against the real observation and the produced passage
|
|
560
|
+
// bytes BEFORE it is ever exposed via promotion -- an unvalidated result is never returned.
|
|
561
|
+
validateGistReceipt({ receipt: sourceReceipt, observation, passagesFile, expectedOwner: observation.owner, mode: 'produce' });
|
|
562
|
+
|
|
563
|
+
fs.mkdirSync(live, { recursive: true });
|
|
564
|
+
promoteArtifactSet({ liveDir: live, candidateDir: stage, files: promoted });
|
|
565
|
+
|
|
566
|
+
const files = promoted.filter((name) => name !== 'RVF-GENERATIONS.json')
|
|
567
|
+
.map((name) => {
|
|
568
|
+
const filePath = path.join(live, name);
|
|
569
|
+
return { name, sha256: crypto.createHash('sha256').update(fs.readFileSync(filePath)).digest('hex'), bytes: fs.statSync(filePath).size };
|
|
570
|
+
});
|
|
571
|
+
return {
|
|
572
|
+
store: 'ruv-gists', kind: 'gist-aggregate', generation,
|
|
573
|
+
sourceMetadata: metadata, files, sourceReceipt, reuseEvidence: captured.reuseEvidence,
|
|
574
|
+
};
|
|
575
|
+
} finally {
|
|
576
|
+
fs.rmSync(stage, { recursive: true, force: true });
|
|
179
577
|
}
|
|
180
|
-
return validateGistReceiptSet(sealGistReceiptSet({ owner: observation.owner, generated: now(),
|
|
181
|
-
observedAt: observation.observedAt, sourceObservationSha256: observation.observationSha256,
|
|
182
|
-
passagesSha256: null, gists }), observation);
|
|
183
578
|
}
|