@clien-ai/mcp 0.10.9 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +152 -0
- package/README.md +33 -10
- package/dist/hypothesis-semantics.js +230 -0
- package/dist/hypothesis-semantics.js.map +1 -0
- package/dist/tools/hypothesis-final.js +55 -0
- package/dist/tools/hypothesis-final.js.map +1 -0
- package/dist/tools/interview-scripts.js +50 -21
- package/dist/tools/interview-scripts.js.map +1 -1
- package/dist/tools/market-sizing-proof.js +69 -4
- package/dist/tools/market-sizing-proof.js.map +1 -1
- package/dist/tools/personas.js +9 -2
- package/dist/tools/personas.js.map +1 -1
- package/dist/tools/receipt-children.js +127 -0
- package/dist/tools/receipt-children.js.map +1 -0
- package/dist/tools/registry.js +49 -11
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/render-safety.js +80 -0
- package/dist/tools/render-safety.js.map +1 -1
- package/dist/tools/report-digest.js +560 -46
- package/dist/tools/report-digest.js.map +1 -1
- package/dist/tools/research.js +28 -3
- package/dist/tools/research.js.map +1 -1
- package/dist/tools/scoped-research.js +40 -11
- package/dist/tools/scoped-research.js.map +1 -1
- package/dist/tools/status.js +46 -2
- package/dist/tools/status.js.map +1 -1
- package/dist/types/report.js +452 -24
- package/dist/types/report.js.map +1 -1
- package/package.json +1 -1
|
@@ -46,8 +46,11 @@
|
|
|
46
46
|
* moves agent-authored text below that separator breaks the digest's own authority
|
|
47
47
|
* rule, which a reader applies positionally.
|
|
48
48
|
*/
|
|
49
|
-
import { collapseWhitespace, safeInline, safeId, clipReceiptQuote, RECEIPT_QUOTE_MAX, } from './render-safety.js';
|
|
49
|
+
import { collapseWhitespace, safeInline, safeId, clipReceiptQuote, containsSpan, hasVisibleContent, RECEIPT_QUOTE_MAX, } from './render-safety.js';
|
|
50
50
|
import { canonicalContentType, contentTypeDeviationLine, contentTypeProvenanceLine, } from './content-type-display.js';
|
|
51
|
+
import { readFinalHypothesisState } from './hypothesis-final.js';
|
|
52
|
+
import { readRobustnessCounts } from '../hypothesis-semantics.js';
|
|
53
|
+
import { resolveSourceReceipt } from './receipt-children.js';
|
|
51
54
|
import { resolveSupportedMarketSizing } from './market-sizing-proof.js';
|
|
52
55
|
/**
|
|
53
56
|
* Max claims rendered per spine. Above this the digest states how many were
|
|
@@ -184,6 +187,34 @@ const UNSOURCED_LABEL = 'unsourced';
|
|
|
184
187
|
const HYPOTHESIS_LABEL = 'hypothesis';
|
|
185
188
|
const SCOPE_CAVEAT_LABEL = '⚠️ scope caveat — persona support from OUTSIDE its credibility domain; directional, NOT grounded';
|
|
186
189
|
const EVIDENCE_READING_LABEL = 'rests on';
|
|
190
|
+
/** FUL-358 — the reliability continuation line's label. */
|
|
191
|
+
const RELIABILITY_LABEL = 'confidence in this verdict';
|
|
192
|
+
/**
|
|
193
|
+
* The marker on a receipt-CHILD line, and the pool sentence that explains it (FUL-685 / T8).
|
|
194
|
+
*
|
|
195
|
+
* ⚠️ IT HAS TO BE VISUALLY SUBORDINATE, for the same reason the pool prints one row per source
|
|
196
|
+
* parent: a child line that opened with `- ` would read as another cached discussion, which is
|
|
197
|
+
* exactly the breadth inflation D13 refuses. `↳` under an existing row reads as "inside this
|
|
198
|
+
* one" and nothing else.
|
|
199
|
+
*
|
|
200
|
+
* ⚠️ AND THE WORD IS `excerpt`, NOT `quote`. T1b (FUL-697) measured the provider transforming
|
|
201
|
+
* the text it returns — deleting a paragraph break, replacing an author's hyperlink with the
|
|
202
|
+
* literal token `URL` — on the rail these children come from. Calling this a speaker's exact
|
|
203
|
+
* words would be a false claim on the one surface whose entire purpose is that the product does
|
|
204
|
+
* not make those. The sentence renders only when a row actually has a child, so a report with
|
|
205
|
+
* no multi-excerpt source is byte-identical to what it rendered before.
|
|
206
|
+
*/
|
|
207
|
+
const RECEIPT_CHILD_MARKER = '↳';
|
|
208
|
+
/**
|
|
209
|
+
* The spine's half of the same fact. One discussion can legitimately yield several excerpts, so
|
|
210
|
+
* a claim that addresses one carries BOTH ids — `sourceId` for the discussion, `receiptId` for
|
|
211
|
+
* the excerpt inside it — and the pool row's matching `↳` line is the text it was checked
|
|
212
|
+
* against. An unresolvable pair is never softened into the parent: it loses the receipt.
|
|
213
|
+
*/
|
|
214
|
+
const RECEIPT_CHILD_SPINE_NOTE = 'A claim carrying a second id after `·` addresses ONE excerpt inside that discussion — match it ' +
|
|
215
|
+
'to the `↳` line of the same id in the receipt pool below, not to the row\'s first excerpt.';
|
|
216
|
+
const RECEIPT_CHILD_POOL_NOTE = 'A `↳` line is the bounded EXCERPT one claim\'s `receiptId` addresses inside that discussion — ' +
|
|
217
|
+
'an excerpt, not a speaker\'s exact words, and the claim line above names the id it points at';
|
|
187
218
|
// ---------------------------------------------------------------------------
|
|
188
219
|
// Defensive accessors — every one of these answers "or nothing" rather than throwing
|
|
189
220
|
// ---------------------------------------------------------------------------
|
|
@@ -221,6 +252,25 @@ function renderId(raw, absent = '(no id)') {
|
|
|
221
252
|
function num(value) {
|
|
222
253
|
return typeof value === 'number' && Number.isFinite(value) ? value : null;
|
|
223
254
|
}
|
|
255
|
+
/**
|
|
256
|
+
* Acceptance is a count, not merely a finite number (FUL-740).
|
|
257
|
+
*
|
|
258
|
+
* Classify before interpreting so an invalid counter can never reach a comparison that reads it
|
|
259
|
+
* as evidence or absence. `Number.isInteger` rejects non-numbers, NaN, Infinity and fractions;
|
|
260
|
+
* the remaining check rejects negative integers. `null` is treated like an omitted field because
|
|
261
|
+
* both mean the producer supplied no usable counter; either way, only `valid` carries a number
|
|
262
|
+
* downstream.
|
|
263
|
+
*/
|
|
264
|
+
function readAcceptanceCounter(value) {
|
|
265
|
+
if (value === undefined || value === null)
|
|
266
|
+
return { kind: 'absent' };
|
|
267
|
+
if (!Number.isInteger(value))
|
|
268
|
+
return { kind: 'invalid' };
|
|
269
|
+
const integer = value;
|
|
270
|
+
if (integer < 0)
|
|
271
|
+
return { kind: 'invalid' };
|
|
272
|
+
return { kind: 'valid', value: integer };
|
|
273
|
+
}
|
|
224
274
|
/**
|
|
225
275
|
* TRI-STATE boolean read: `true`, `false`, or `null` for "absent or unreadable".
|
|
226
276
|
*
|
|
@@ -272,7 +322,27 @@ function safeReceiptUrl(raw) {
|
|
|
272
322
|
return null;
|
|
273
323
|
}
|
|
274
324
|
}
|
|
275
|
-
/**
|
|
325
|
+
/**
|
|
326
|
+
* Resolve the single, persona-owned presentation grant behind a GROUNDED claim.
|
|
327
|
+
*
|
|
328
|
+
* ⚠️ FUL-711 — TWO REFERENCES, AND THEY MUST AGREE. `sourceId` locates the source parent
|
|
329
|
+
* positionally; an optional `receiptId` locates the exact excerpt CHILD inside it. The three
|
|
330
|
+
* outcomes are deliberately not two:
|
|
331
|
+
*
|
|
332
|
+
* - **no `receiptId` key (or raw `undefined`)** — the legacy path. Resolve the parent, render
|
|
333
|
+
* its single cached excerpt, exactly as before this function learned about children.
|
|
334
|
+
* - **`receiptId` resolves AND the child carries the claim's `quoteSpan`** — the claim
|
|
335
|
+
* addresses that child, and the child's own text is what `clipReceiptQuote` gets as its
|
|
336
|
+
* haystack. Both halves, or neither: FUL-728 added the span check because resolving to a
|
|
337
|
+
* child that does not contain the span is FUL-711's defect with a content-addressed id on
|
|
338
|
+
* it (see the call site).
|
|
339
|
+
* - **`receiptId` present but unresolvable** — `null`, malformed, dangling, cross-source,
|
|
340
|
+
* duplicate, or a parent with no children at all: **return `null` and fail closed.** Our
|
|
341
|
+
* producer omits this optional key, so an explicit `null` has unknown provenance and cannot
|
|
342
|
+
* claim the legacy parent's quote as grounded evidence. Falling back would print a 200-char
|
|
343
|
+
* clip of excerpt 1 under a GROUNDED badge for a claim grounded on excerpt 2 — precisely the
|
|
344
|
+
* defect this resolution exists to remove, restored as an error path.
|
|
345
|
+
*/
|
|
276
346
|
function resolvePersonaReceipt(rawClaim, personas) {
|
|
277
347
|
const claim = asRecord(rawClaim);
|
|
278
348
|
if (str(claim?.state) !== 'GROUNDED')
|
|
@@ -288,11 +358,71 @@ function resolvePersonaReceipt(rawClaim, personas) {
|
|
|
288
358
|
const href = safeReceiptUrl(source?.url);
|
|
289
359
|
if (!source || !href)
|
|
290
360
|
return null;
|
|
361
|
+
// Branch on the RAW value, not on the resolved one: only `undefined` is "legacy claim, nothing
|
|
362
|
+
// to resolve" and every present value — including `null` — is a promise this parent must keep.
|
|
363
|
+
const claimedReceiptId = claim?.receiptId;
|
|
364
|
+
let child = null;
|
|
365
|
+
if (claimedReceiptId !== undefined) {
|
|
366
|
+
child = resolveSourceReceipt(source.receipts, claimedReceiptId) ?? null;
|
|
367
|
+
if (!child)
|
|
368
|
+
return null;
|
|
369
|
+
// ⚠️ RESOLVING THE ID IS NOT CHECKING THE SPAN, AND FUL-711'S DEFECT SURVIVES THE GAP
|
|
370
|
+
// (FUL-728). A `receiptId` that is well-formed, lives under the RIGHT parent, and is unique
|
|
371
|
+
// there resolves cleanly — while the claim's `quoteSpan` sits in a DIFFERENT child of that
|
|
372
|
+
// same parent. Nothing above notices: the id was never asked to agree with the span, only to
|
|
373
|
+
// exist. `clipReceiptQuote` then finds no span in the child it was handed, takes the head
|
|
374
|
+
// clip, and the digest prints `[GROUNDED]` beside an excerpt that does not contain the
|
|
375
|
+
// sentence which earned the badge — FUL-711 resolved to the wrong child instead of
|
|
376
|
+
// positionally, which is the same lie with a content-addressed id on it. D12 asks the
|
|
377
|
+
// selected receipt to CONTAIN the exact span, so ask it here.
|
|
378
|
+
//
|
|
379
|
+
// ⚠️ THIS IS A DIFFERENT AXIS FROM THE `> 1` BOUND BELOW, and neither substitutes for the
|
|
380
|
+
// other. That one asks whether a claim named a child at all; this one asks whether the child
|
|
381
|
+
// it named is the right one. Widening that bound would break the single-child byte-identity
|
|
382
|
+
// guarantee it exists to hold; this check cannot, because it lives entirely inside the branch
|
|
383
|
+
// a `receiptId` opens — a legacy claim carries none and never reaches it.
|
|
384
|
+
//
|
|
385
|
+
// ⚠️ NO `quoteSpan` STILL GRANTS. There is no span to contradict the child, and refusing on
|
|
386
|
+
// absence would be a different rule — "a Reddit claim must carry a span" — which belongs to
|
|
387
|
+
// the producer and to T5's schema, not to a renderer inferring it. `containsSpan` uses the
|
|
388
|
+
// renderer's OWN lookup, so a span this package could not locate can never earn a badge here
|
|
389
|
+
// and then quietly render as an unwindowed head clip.
|
|
390
|
+
//
|
|
391
|
+
// ⚠️ BRANCH ON THE RAW VALUE, exactly as `receiptId` does two lines up (#1059 review). `str()`
|
|
392
|
+
// answers `null` for a NUMBER, an OBJECT and `''` alike, so reading through it would file a
|
|
393
|
+
// present-but-malformed span under "legacy claim, nothing to check" and grant. Those are
|
|
394
|
+
// different facts: absence is a claim that never promised a span, while a present one is a
|
|
395
|
+
// promise this child has to keep, and a producer that emitted `42` or `''` there has told us
|
|
396
|
+
// nothing about which sentence earned the badge. Fail closed on the promise it cannot keep.
|
|
397
|
+
const rawSpan = claim?.quoteSpan;
|
|
398
|
+
if (rawSpan !== undefined && rawSpan !== null) {
|
|
399
|
+
const span = str(rawSpan);
|
|
400
|
+
if (span === null || !containsSpan(child.excerpt, span))
|
|
401
|
+
return null;
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
else if (asArray(source.receipts).length > 1) {
|
|
405
|
+
// ⚠️ ABSENT `receiptId` ON A parent carrying SEVERAL children is not a legacy claim — it is a
|
|
406
|
+
// producer that failed to name one, and it is the only case that reaches the fallback this
|
|
407
|
+
// function's docblock forbids. `quote` holds excerpt 1 (D13: one staging row per permalink),
|
|
408
|
+
// so returning the parent prints a clip of excerpt 1 under a GROUNDED badge for a claim whose
|
|
409
|
+
// span may live in excerpt 2 — FUL-711's exact defect, restored as an error path that fires
|
|
410
|
+
// only on malformed input nobody is watching.
|
|
411
|
+
//
|
|
412
|
+
// ⚠️ THE BOUND IS `> 1`, NOT `> 0`, AND THAT IS D15's LINE — "multi-receipt evidence" needs an
|
|
413
|
+
// addressed child; "legacy single-receipt" keeps positional behaviour. With exactly ONE child
|
|
414
|
+
// the parent row and that child carry the same text BY CONSTRUCTION, so resolving positionally
|
|
415
|
+
// cannot show a span-free excerpt and the legacy byte-identity guarantee holds. Widening this
|
|
416
|
+
// to `> 0` breaks that guarantee for one-child sources, which is a real regression: an
|
|
417
|
+
// additive field would start changing what a legacy reader sees.
|
|
418
|
+
return null;
|
|
419
|
+
}
|
|
291
420
|
return {
|
|
292
421
|
sourceId: `RCP-p${personaIndex}-s${sourceIndex}`,
|
|
293
422
|
href,
|
|
294
423
|
quote: collapseWhitespace(str(source.quote) ?? ''),
|
|
295
424
|
source,
|
|
425
|
+
child,
|
|
296
426
|
};
|
|
297
427
|
}
|
|
298
428
|
function resolvePersonaRows(reportData) {
|
|
@@ -369,7 +499,13 @@ function renderPersonaClaim(raw, resolvedReceipt) {
|
|
|
369
499
|
// why a pointer on a non-GROUNDED claim must never render as a receipt.
|
|
370
500
|
let provenance = '';
|
|
371
501
|
if (resolvedReceipt) {
|
|
372
|
-
|
|
502
|
+
// FUL-685: when the claim addressed a receipt CHILD, the arrow carries both halves of the
|
|
503
|
+
// pair. Without the child id a reader looking at two claims on one `RCP-p0-s0` row cannot
|
|
504
|
+
// tell which `↳` excerpt grounds which claim — the same "wrong excerpt under a GROUNDED
|
|
505
|
+
// badge" confusion this change removes, relocated one line down. `safeId` because it is an
|
|
506
|
+
// identifier: exact or nothing.
|
|
507
|
+
const childId = resolvedReceipt.child ? safeId(resolvedReceipt.child.receiptId) : null;
|
|
508
|
+
provenance = ` ← ${resolvedReceipt.sourceId}${childId ? ` · ${childId}` : ''}`;
|
|
373
509
|
}
|
|
374
510
|
else if (rawState === 'GROUNDED') {
|
|
375
511
|
provenance = ' · Stored grade: GROUNDED — receipt unavailable; do not cite';
|
|
@@ -387,7 +523,8 @@ function renderPersonaClaim(raw, resolvedReceipt) {
|
|
|
387
523
|
* A stored GROUNDED grade and a model-attestation marker are both untrusted report JSON, not proof
|
|
388
524
|
* by themselves. Either earns its presentation only when its canonical pointer indexes the exact
|
|
389
525
|
* `reportEvidence` pool and the target has a non-credentialed HTTP(S) destination. An attestation
|
|
390
|
-
* additionally needs
|
|
526
|
+
* additionally needs a captured window with something VISIBLE in it — not merely a nonblank
|
|
527
|
+
* string, which U+200B and the C0 controls satisfy. Keeping this as one resolver lets the row,
|
|
391
528
|
* tally, pointer and window consume the same answer instead of minting orphan citations.
|
|
392
529
|
*/
|
|
393
530
|
function resolveReportReceipt(rawClaim, reportEvidence) {
|
|
@@ -405,7 +542,13 @@ function resolveReportReceipt(rawClaim, reportEvidence) {
|
|
|
405
542
|
const attestation = asRecord(claim?.attestation);
|
|
406
543
|
sourceId = typeof attestation?.sourceId === 'string' ? attestation.sourceId.trim() : '';
|
|
407
544
|
quote = typeof attestation?.quote === 'string' ? collapseWhitespace(attestation.quote) : '';
|
|
408
|
-
|
|
545
|
+
// ⚠️ NONBLANK IS NOT THE SAME TEST AS "HAS A WINDOW" (#1059 review, same defect as FUL-728
|
|
546
|
+
// item 3). `collapseWhitespace` only collapses `\s`, which covers none of U+200B, U+2060,
|
|
547
|
+
// U+00AD or the C0 controls — so an attestation window built from those survives this as a
|
|
548
|
+
// non-empty string, clears the guard, and the claim earns `cited → RRCP-s0` plus an
|
|
549
|
+
// `Attested window: ""` line. A citation marker over a window nobody can read is the
|
|
550
|
+
// presentation this resolver exists to withhold.
|
|
551
|
+
if (!hasVisibleContent(quote))
|
|
409
552
|
return null;
|
|
410
553
|
}
|
|
411
554
|
else {
|
|
@@ -523,9 +666,14 @@ function renderPersonaSpine(reportData, channel) {
|
|
|
523
666
|
const tally = tallyPersonaStates(rows);
|
|
524
667
|
const shown = rows.slice(0, CLAIM_RENDER_CAP);
|
|
525
668
|
const lines = shown.map(({ claim, receipt }) => renderPersonaClaim(claim, receipt));
|
|
669
|
+
// FUL-685: only say the pair exists when a rendered claim actually carries one. A report with
|
|
670
|
+
// no multi-excerpt source renders this header byte-identically to before.
|
|
671
|
+
const anyChild = shown.some(({ receipt }) => receipt?.child);
|
|
526
672
|
return (`### Persona claim spine — ${tallyLine(tally, rows.length)}\n` +
|
|
527
673
|
'A GROUNDED claim\'s receipt id (`RCP-p{i}-s{j}`) indexes `personas[i].sources[j]` — the cached ' +
|
|
528
|
-
'forum post its quote span was code-verified against.
|
|
674
|
+
'forum post its quote span was code-verified against. ' +
|
|
675
|
+
(anyChild ? `${RECEIPT_CHILD_SPINE_NOTE} ` : '') +
|
|
676
|
+
'A stored GROUNDED grade whose canonical, ' +
|
|
529
677
|
'persona-owned safe receipt is unavailable is rendered and counted NO_RECEIPT. NO_RECEIPT can mean a bad claim OR merely ' +
|
|
530
678
|
'sparse evidence: check that persona\'s `insufficientEvidence` / `sourcesFound` below before ' +
|
|
531
679
|
'discounting it. Anything not GROUNDED is unverified.\n' +
|
|
@@ -550,7 +698,26 @@ function renderReportSpine(reportData, channel) {
|
|
|
550
698
|
const attestedNote = attested > 0 ? ` (of which ${attested} ${ATTESTED_LABEL})` : '';
|
|
551
699
|
const shown = rows.slice(0, CLAIM_RENDER_CAP);
|
|
552
700
|
const lines = shown.map(({ claim, receipt }) => renderReportClaim(claim, receipt));
|
|
553
|
-
|
|
701
|
+
// FUL-358 — SAY WHICH POPULATION THE HEADLINE COUNTED.
|
|
702
|
+
//
|
|
703
|
+
// The prose below has always explained that a `summary` claim usually restates a `market` or
|
|
704
|
+
// `competitor` one and "must not be counted as a second independent finding" — while the
|
|
705
|
+
// headline directly above it counted exactly that. The report markdown in the SAME tool
|
|
706
|
+
// result scopes its own tally to the market+competitor spine and names it
|
|
707
|
+
// (`of N market & competitor claims`), so the two numbers arrived side by side looking like
|
|
708
|
+
// one of them was an arithmetic error. Both are defensible; only one said what it counted.
|
|
709
|
+
//
|
|
710
|
+
// The tally still covers every claim — this section is the audit surface, and a graded
|
|
711
|
+
// summary claim must be visible in it. What is added is the split, so a reader can reconcile
|
|
712
|
+
// the two headlines instead of choosing between them.
|
|
713
|
+
const summaryClaims = rows.filter(({ claim }) => asRecord(claim)?.section === 'summary').length;
|
|
714
|
+
const spineClaims = claims.length - summaryClaims;
|
|
715
|
+
const populationNote = `\nDENOMINATOR: ${claims.length} = ${spineClaims} market & competitor claim(s) + ` +
|
|
716
|
+
`${summaryClaims} executive-summary restatement(s). The report markdown above tallies the ` +
|
|
717
|
+
`${spineClaims} market & competitor claims ONLY, and names that scope; this section tallies ` +
|
|
718
|
+
'all three sections because it is the audit surface. The two headlines are the same run ' +
|
|
719
|
+
'counted over two populations, not a discrepancy.';
|
|
720
|
+
return (`### Report claim spine — ${tallyLine(tally, claims.length)}${attestedNote}${populationNote}\n` +
|
|
554
721
|
'A SEPARATE pool from the persona spine — the two never cross. Sections are `market` (sizing/' +
|
|
555
722
|
'trend figures), `competitor` (profile facts) and `summary` (an assertion quoted VERBATIM from ' +
|
|
556
723
|
'the executive summary, graded against the same evidence — so a grounded `summary` claim ' +
|
|
@@ -582,6 +749,25 @@ function renderReportSpine(reportData, channel) {
|
|
|
582
749
|
* has no visible badge, so windowing a quote to support it would spend the excerpt
|
|
583
750
|
* on a claim the agent cannot see, at the cost of one it can.
|
|
584
751
|
*/
|
|
752
|
+
/**
|
|
753
|
+
* The bucket a claim's verified span belongs to.
|
|
754
|
+
*
|
|
755
|
+
* ⚠️ FUL-711 — KEYED BY THE HAYSTACK, NOT BY THE SOURCE. Before receipt children there was one
|
|
756
|
+
* text per source and `sourceId` was both. Now a source can carry several excerpts, and a span
|
|
757
|
+
* verified inside excerpt 2 is not findable in excerpt 1 — so windowing the parent by it would
|
|
758
|
+
* locate nothing and silently fall back to the head clip. One bucket per (parent, child), with
|
|
759
|
+
* the childless bucket keyed by the empty string, keeps every span pointed at the text it is
|
|
760
|
+
* actually in. A NUL (`U+0000`) cannot appear in either id, so the two halves cannot alias.
|
|
761
|
+
*
|
|
762
|
+
* ⚠️ THE SEPARATOR IS WRITTEN `\u0000`, NEVER AS THE RAW BYTE. A literal NUL anywhere in a
|
|
763
|
+
* `.ts` file makes `file` classify it as `data` and makes grep and ripgrep skip it with a bare
|
|
764
|
+
* `Binary file matches` — which, on the digest, silently removes the repo's most trust-critical
|
|
765
|
+
* renderer from every search an agent runs here. The escape is the same byte at runtime.
|
|
766
|
+
* Enforced by `__tests__/integration/workflows/source-files-stay-greppable.test.ts`.
|
|
767
|
+
*/
|
|
768
|
+
function receiptBucketKey(sourceId, receiptId) {
|
|
769
|
+
return `${sourceId}\u0000${receiptId ?? ''}`;
|
|
770
|
+
}
|
|
585
771
|
function collectVerifiedSpans(rows) {
|
|
586
772
|
const spans = new Map();
|
|
587
773
|
for (const { claim: raw, receipt } of rows.slice(0, CLAIM_RENDER_CAP)) {
|
|
@@ -591,11 +777,12 @@ function collectVerifiedSpans(rows) {
|
|
|
591
777
|
const span = str(claim.quoteSpan);
|
|
592
778
|
if (!span)
|
|
593
779
|
continue;
|
|
594
|
-
const
|
|
780
|
+
const key = receiptBucketKey(receipt.sourceId, receipt.child?.receiptId ?? null);
|
|
781
|
+
const existing = spans.get(key);
|
|
595
782
|
if (existing)
|
|
596
783
|
existing.push(span);
|
|
597
784
|
else
|
|
598
|
-
spans.set(
|
|
785
|
+
spans.set(key, [span]);
|
|
599
786
|
}
|
|
600
787
|
return spans;
|
|
601
788
|
}
|
|
@@ -622,18 +809,36 @@ function collectVerifiedSpans(rows) {
|
|
|
622
809
|
* pages, whose "quote" would be a page excerpt chosen at fetch time rather than a
|
|
623
810
|
* human's own words, and the pool already carries the one thing a reader needs
|
|
624
811
|
* from them — `publishedDate`, the figure's actual recency (FUL-148).
|
|
812
|
+
*
|
|
813
|
+
* FUL-685 (T8 / D13/D15), spec in FUL-711: a source whose ONE url yielded several bounded
|
|
814
|
+
* excerpts gets ONE row here, still — receipt count must never inflate source or discussion
|
|
815
|
+
* breadth — with an indented `↳` line per excerpt a rendered claim actually ADDRESSES. Not per
|
|
816
|
+
* stored excerpt: printing the whole child set under a row would make one discussion look like
|
|
817
|
+
* several, which is the inflation D13 exists to prevent, and would spend the digest budget on
|
|
818
|
+
* text no claim points at.
|
|
625
819
|
*/
|
|
626
820
|
function renderReceiptPools(reportData, channel) {
|
|
627
821
|
const reportEvidence = asArray(reportData.reportEvidence);
|
|
628
822
|
const personaRows = resolvePersonaRows(reportData).slice(0, CLAIM_RENDER_CAP);
|
|
629
823
|
const spansByReceipt = collectVerifiedSpans(personaRows);
|
|
824
|
+
/**
|
|
825
|
+
* One entry per source PARENT, carrying the receipt children the rendered claims addressed.
|
|
826
|
+
* A `Map` keyed by child id so two claims on the same excerpt collapse to one `↳` line.
|
|
827
|
+
*/
|
|
630
828
|
const personaReceiptById = new Map();
|
|
631
829
|
for (const { receipt } of personaRows) {
|
|
632
|
-
if (receipt)
|
|
633
|
-
|
|
830
|
+
if (!receipt)
|
|
831
|
+
continue;
|
|
832
|
+
let row = personaReceiptById.get(receipt.sourceId);
|
|
833
|
+
if (!row) {
|
|
834
|
+
row = { receipt, children: new Map() };
|
|
835
|
+
personaReceiptById.set(receipt.sourceId, row);
|
|
836
|
+
}
|
|
837
|
+
if (receipt.child)
|
|
838
|
+
row.children.set(receipt.child.receiptId, receipt.child);
|
|
634
839
|
}
|
|
635
840
|
const personaReceipts = [];
|
|
636
|
-
for (const [receiptId, receipt] of personaReceiptById) {
|
|
841
|
+
for (const [receiptId, { receipt, children }] of personaReceiptById) {
|
|
637
842
|
const source = receipt.source;
|
|
638
843
|
// FUL-253: the quote below was already flattened — these two were not, and
|
|
639
844
|
// they sit on the SAME `- RCP-…` line, so a newline in either forges the
|
|
@@ -647,19 +852,59 @@ function renderReceiptPools(reportData, channel) {
|
|
|
647
852
|
// `collapseWhitespace` is a FORGERY GUARD, not formatting: this pool is a
|
|
648
853
|
// list of `- RCP-i-j — …` lines, and a quote containing a newline plus a
|
|
649
854
|
// convincing `- RCP-` prefix would add an entry pointing at a source nobody
|
|
650
|
-
// retrieved.
|
|
651
|
-
//
|
|
855
|
+
// retrieved.
|
|
856
|
+
//
|
|
857
|
+
// ⚠️ WHETHER TO PRINT THE LINE AT ALL IS A DIFFERENT QUESTION, AND `hasVisibleContent` IS THE
|
|
858
|
+
// ONE THAT ANSWERS IT (#1059 review, same defect as FUL-728 item 3). A whitespace-only quote
|
|
859
|
+
// collapses to '' and correctly renders as a bare pointer — "the source said nothing" — but
|
|
860
|
+
// a quote of U+200B or U+0000 survives the collapse non-empty, and this line then prints `""`
|
|
861
|
+
// under a receipt GROUNDED claims point at. The forgery guard stays on the flattening; the
|
|
862
|
+
// emptiness question is asked of the visible characters.
|
|
652
863
|
//
|
|
653
864
|
// FUL-252: WHICH `RECEIPT_QUOTE_MAX` characters is chosen by the spans of the
|
|
654
865
|
// GROUNDED claims pointing at THIS receipt, not by the quote's head. Passing
|
|
655
866
|
// the receipt's own id is the whole wiring — `RCP-p{i}-s{j}` is the pointer
|
|
656
867
|
// `renderPersonaClaim` prints, so the two sides of the arrow are built from
|
|
657
868
|
// the same expression and cannot drift into windowing the wrong receipt.
|
|
869
|
+
//
|
|
870
|
+
// ⚠️ FUL-685: the parent bucket is now the CHILDLESS one. A span verified inside excerpt 2
|
|
871
|
+
// is not findable in excerpt 1, so windowing this line by it would locate nothing and fall
|
|
872
|
+
// back to the head clip while looking like a window. Legacy claims carry no `receiptId`, so
|
|
873
|
+
// for every pre-existing report every span is still in this bucket and this line is
|
|
874
|
+
// byte-identical to what it rendered before.
|
|
658
875
|
const quote = receipt.quote;
|
|
659
|
-
const clipped = clipReceiptQuote(quote, spansByReceipt.get(receiptId) ?? [], RECEIPT_QUOTE_MAX);
|
|
876
|
+
const clipped = clipReceiptQuote(quote, spansByReceipt.get(receiptBucketKey(receiptId, null)) ?? [], RECEIPT_QUOTE_MAX);
|
|
660
877
|
const topic = safeInline(source?.topic, 40);
|
|
661
878
|
const topicLine = topic ? `\n On ${topic}` : '';
|
|
662
|
-
const quoteLine = quote ? `\n "${clipped}"` : '';
|
|
879
|
+
const quoteLine = hasVisibleContent(quote) ? `\n "${clipped}"` : '';
|
|
880
|
+
// The addressed receipt CHILDREN, one line each, sorted by receipt id.
|
|
881
|
+
//
|
|
882
|
+
// ⚠️ THE PARENT LINE STAYS, EVEN WHEN A CHILD REPEATS IT. The parent's `quote` IS the first
|
|
883
|
+
// excerpt (that is what `admitRedditPersonaSources` writes), so a claim addressing child 1
|
|
884
|
+
// prints the same sentence twice. That redundancy is deliberate and cheaper than the
|
|
885
|
+
// alternatives: suppressing the parent line whenever every claim addressed a child means a
|
|
886
|
+
// row can render with no excerpt at all on any path that miscounts, and suppressing a `↳`
|
|
887
|
+
// line that merely duplicates the parent breaks the header's own instruction — a claim's id
|
|
888
|
+
// would have no line to match, which is the mapping this change exists to establish.
|
|
889
|
+
//
|
|
890
|
+
// ⚠️ SORTED BY ID, NOT BY CLAIM ORDER. The id is content-addressed — `deriveReceiptId`
|
|
891
|
+
// hashes (normalized url, exact excerpt) — so sorting on it is stable across retry, merge,
|
|
892
|
+
// reopen and promotion, and matches the order the producer emits children in. Ordering by
|
|
893
|
+
// first-claim-seen would instead make the pool's shape a function of how the synthesiser
|
|
894
|
+
// happened to sequence its claims.
|
|
895
|
+
//
|
|
896
|
+
// `safeId` rather than `safeInline`: this is an IDENTIFIER, and a repaired id is a wrong id
|
|
897
|
+
// that still reads as one. It cannot fail in practice — nothing reaches here without
|
|
898
|
+
// matching `rr{n}:{16 hex}` — but exact-or-nothing is the rule for every id in this package
|
|
899
|
+
// and a silently clipped one would be a pointer to a child that does not exist.
|
|
900
|
+
const childLines = [...children.values()]
|
|
901
|
+
.sort((a, b) => (a.receiptId < b.receiptId ? -1 : a.receiptId > b.receiptId ? 1 : 0))
|
|
902
|
+
.map((child) => {
|
|
903
|
+
const childId = safeId(child.receiptId) ?? '(unrenderable receipt id)';
|
|
904
|
+
const clippedChild = clipReceiptQuote(child.excerpt, spansByReceipt.get(receiptBucketKey(receiptId, child.receiptId)) ?? [], RECEIPT_QUOTE_MAX);
|
|
905
|
+
return `\n ${RECEIPT_CHILD_MARKER} ${childId} — "${clippedChild}"`;
|
|
906
|
+
})
|
|
907
|
+
.join('');
|
|
663
908
|
// FUL-560: the source type, stated ONCE for the pool (in the header below) and inline only
|
|
664
909
|
// on a row that breaks the stated rule. This pool is the one place in the package where the
|
|
665
910
|
// type is a near-constant: `stampEvidenceMetadata` (agent-side) prunes every non-`user_voice`
|
|
@@ -673,8 +918,9 @@ function renderReceiptPools(reportData, channel) {
|
|
|
673
918
|
// would be exactly the "silence reads as reassurance" fail-open this change exists to close.
|
|
674
919
|
const personaType = canonicalContentType(source?.contentType);
|
|
675
920
|
const deviationLine = personaType === 'user_voice' ? '' : contentTypeDeviationLine(source?.contentType);
|
|
676
|
-
personaReceipts.push(`- ${receiptId} — ${platform} — ${url}${topicLine}${quoteLine}${deviationLine}`);
|
|
921
|
+
personaReceipts.push(`- ${receiptId} — ${platform} — ${url}${topicLine}${quoteLine}${childLines}${deviationLine}`);
|
|
677
922
|
}
|
|
923
|
+
const anyReceiptChildren = [...personaReceiptById.values()].some((row) => row.children.size > 0);
|
|
678
924
|
const evidenceReceipts = reportEvidence.map((raw, n) => {
|
|
679
925
|
const source = asRecord(raw);
|
|
680
926
|
// FUL-253: every field on this row is scraped-page metadata, and the row is
|
|
@@ -713,7 +959,9 @@ function renderReceiptPools(reportData, channel) {
|
|
|
713
959
|
if (personaReceipts.length > 0) {
|
|
714
960
|
const shown = personaReceipts.slice(0, CLAIM_RENDER_CAP);
|
|
715
961
|
sections.push(`**Persona receipts** (${plural(personaReceipts.length, 'cached forum post')}) — safe targets of the visible \`RCP-\` pointers above. ` +
|
|
716
|
-
'Every receipt here is a first-hand `user_voice` post unless its own row says otherwise
|
|
962
|
+
'Every receipt here is a first-hand `user_voice` post unless its own row says otherwise' +
|
|
963
|
+
(anyReceiptChildren ? `. ${RECEIPT_CHILD_POOL_NOTE}` : '') +
|
|
964
|
+
':\n' +
|
|
717
965
|
shown.join('\n') +
|
|
718
966
|
capNote(shown.length, personaReceipts.length, channel, 'report_data.personas[].sources'));
|
|
719
967
|
}
|
|
@@ -902,18 +1150,61 @@ function renderSycophancy(reportData, channel) {
|
|
|
902
1150
|
* FUL-263 — before anything rendered them, which is that file's whole premise — so its
|
|
903
1151
|
* assertions bit on this change with no fixture edit.
|
|
904
1152
|
*/
|
|
905
|
-
function hypothesisDetail(result) {
|
|
1153
|
+
function hypothesisDetail(result, includeScopeCaveat = true) {
|
|
906
1154
|
const statement = safeInline(result?.statement, HYPOTHESIS_STATEMENT_MAX);
|
|
907
1155
|
const evidenceReading = safeInline(result?.evidenceReading, HYPOTHESIS_NOTE_MAX);
|
|
908
1156
|
return ((statement ? `\n ${HYPOTHESIS_LABEL}: "${statement}"` : '') +
|
|
909
|
-
scopeCaveatDetail(result) +
|
|
1157
|
+
(includeScopeCaveat ? scopeCaveatDetail(result) : '') +
|
|
1158
|
+
reliabilityDetail(result) +
|
|
910
1159
|
(evidenceReading ? `\n ${EVIDENCE_READING_LABEL}: ${evidenceReading}` : ''));
|
|
911
1160
|
}
|
|
1161
|
+
/**
|
|
1162
|
+
* The FUL-358 reliability line — how much to trust this verdict, and how much evidence is
|
|
1163
|
+
* behind it.
|
|
1164
|
+
*
|
|
1165
|
+
* ## Why this section renders a reliability at all, having settled not to render `confidence`
|
|
1166
|
+
*
|
|
1167
|
+
* FUL-614 deliberately withheld the numeric `confidence`: it was model-assigned, and printing
|
|
1168
|
+
* it beside the code-derived survival count invited a reader to average two things that are not
|
|
1169
|
+
* comparable. That reasoning holds and this does not contradict it — `final.confidence` is
|
|
1170
|
+
* itself CODE-DERIVED, from the evidence the row carries and from what these very re-asks did.
|
|
1171
|
+
* It is the same KIND of value as the survival count beside it, which is exactly what the
|
|
1172
|
+
* withheld number was not. The retired numeric field stays withheld.
|
|
1173
|
+
*
|
|
1174
|
+
* Values come from a closed enum, so they cannot forge a line and need no `safeInline`; an
|
|
1175
|
+
* unrecognised value emits nothing rather than being quoted through. A legacy row has no block
|
|
1176
|
+
* and emits nothing — its numeric `confidence` is a different measure and is not restated here.
|
|
1177
|
+
*/
|
|
1178
|
+
function reliabilityDetail(result) {
|
|
1179
|
+
const final = readFinalHypothesisState(result);
|
|
1180
|
+
if (!final.fromContract)
|
|
1181
|
+
return '';
|
|
1182
|
+
const { confidence, sufficiency, confidenceBasis: basis } = final;
|
|
1183
|
+
const reliability = confidence
|
|
1184
|
+
? `\`${confidence}\``
|
|
1185
|
+
: basis === 'withheld_no_evidence'
|
|
1186
|
+
? 'WITHHELD — the run attached no evidence to this hypothesis'
|
|
1187
|
+
: null;
|
|
1188
|
+
if (!reliability && !sufficiency)
|
|
1189
|
+
return '';
|
|
1190
|
+
const parts = [
|
|
1191
|
+
reliability ? `${RELIABILITY_LABEL}: ${reliability}` : null,
|
|
1192
|
+
sufficiency ? `evidence ${sufficiency}` : null,
|
|
1193
|
+
].filter(Boolean);
|
|
1194
|
+
return `\n ${parts.join('; ')} (reliability of the verdict, NOT the probability the hypothesis is true)`;
|
|
1195
|
+
}
|
|
912
1196
|
/** The exact quoted/labelled FUL-614 shape, shared by robust and caveat-only rows. */
|
|
913
1197
|
function scopeCaveatDetail(result) {
|
|
914
1198
|
const scopeCaveat = safeInline(result?.scopeCaveat, HYPOTHESIS_NOTE_MAX);
|
|
915
1199
|
return scopeCaveat ? `\n ${SCOPE_CAVEAT_LABEL}: "${scopeCaveat}"` : '';
|
|
916
1200
|
}
|
|
1201
|
+
/** A present, non-null value records an attempted re-test even when its shape is unreadable. */
|
|
1202
|
+
function hasRawRobustnessAttempt(result) {
|
|
1203
|
+
return (result !== null &&
|
|
1204
|
+
Object.prototype.hasOwnProperty.call(result, 'robustness') &&
|
|
1205
|
+
result.robustness !== null &&
|
|
1206
|
+
result.robustness !== undefined);
|
|
1207
|
+
}
|
|
917
1208
|
/**
|
|
918
1209
|
* Scope caveats for hypotheses that do not have a robustness row (FUL-626).
|
|
919
1210
|
*
|
|
@@ -932,7 +1223,7 @@ function scopeCaveatDetail(result) {
|
|
|
932
1223
|
function renderUnretestedScopeCaveats(reportData) {
|
|
933
1224
|
const lines = asArray(reportData.hypothesisResults).flatMap((raw) => {
|
|
934
1225
|
const result = asRecord(raw);
|
|
935
|
-
if (!result ||
|
|
1226
|
+
if (!result || hasRawRobustnessAttempt(result))
|
|
936
1227
|
return [];
|
|
937
1228
|
const detail = scopeCaveatDetail(result);
|
|
938
1229
|
if (!detail)
|
|
@@ -1115,13 +1406,16 @@ function renderRobustness(reportData, channel) {
|
|
|
1115
1406
|
const results = asArray(reportData.hypothesisResults);
|
|
1116
1407
|
if (results.length === 0)
|
|
1117
1408
|
return '';
|
|
1118
|
-
const
|
|
1119
|
-
|
|
1409
|
+
const renderable = results.filter((raw) => {
|
|
1410
|
+
const result = asRecord(raw);
|
|
1411
|
+
return hasRawRobustnessAttempt(result) || readFinalHypothesisState(result).fromContract;
|
|
1412
|
+
});
|
|
1413
|
+
if (renderable.length === 0) {
|
|
1120
1414
|
return ('### Hypothesis robustness — NOT RE-TESTED on this run\n' +
|
|
1121
1415
|
'No verdict was re-asked under a reworded framing, so no verdict above carries a survival ' +
|
|
1122
1416
|
'count. Treat each as a single-shot answer.');
|
|
1123
1417
|
}
|
|
1124
|
-
const shown =
|
|
1418
|
+
const shown = renderable.slice(0, LIST_RENDER_CAP);
|
|
1125
1419
|
const lines = shown.map((raw) => {
|
|
1126
1420
|
const result = asRecord(raw);
|
|
1127
1421
|
// FUL-253: a `- ` row whose whole point is the ⚠️ FLIPPED warning. A newline
|
|
@@ -1131,43 +1425,62 @@ function renderRobustness(reportData, channel) {
|
|
|
1131
1425
|
const id = renderId(result?.hypothesisId);
|
|
1132
1426
|
const status = safeInline(result?.status, 40) ?? 'unknown';
|
|
1133
1427
|
const robustness = asRecord(result?.robustness);
|
|
1134
|
-
const
|
|
1135
|
-
const
|
|
1136
|
-
const
|
|
1137
|
-
const
|
|
1138
|
-
const count =
|
|
1428
|
+
const counts = readRobustnessCounts(robustness);
|
|
1429
|
+
const reading = readFinalHypothesisState(result);
|
|
1430
|
+
const outcome = reading.robustness;
|
|
1431
|
+
const effectiveVerdict = reading.verdict ?? status;
|
|
1432
|
+
const count = counts ? `held ${counts.survived}/${counts.total} framings` : 'survival count unavailable';
|
|
1139
1433
|
// Appended to EVERY branch, not just the flipped one. A held verdict is the row a
|
|
1140
1434
|
// reader is most likely to act on unexamined, so it is the last one that should be
|
|
1141
1435
|
// identified by an id alone.
|
|
1142
|
-
|
|
1143
|
-
|
|
1436
|
+
// A single-shot row's caveat stays in the dedicated un-retested-caveats block below rather
|
|
1437
|
+
// than being duplicated here. Its statement, reliability and evidence reading still belong
|
|
1438
|
+
// on this final-state row.
|
|
1439
|
+
const detail = hypothesisDetail(result, outcome !== 'not_retested');
|
|
1440
|
+
if (outcome === 'not_retested') {
|
|
1441
|
+
return `- ${id}: final verdict \`${effectiveVerdict}\` — NOT RE-TESTED under rephrasing; treat this verdict as a single-shot answer.${detail}`;
|
|
1442
|
+
}
|
|
1443
|
+
if (outcome === 'flipped') {
|
|
1444
|
+
const verdictInstruction = effectiveVerdict === status
|
|
1445
|
+
? `The final verdict remains \`${effectiveVerdict}\`; the flip lowers its reliability.`
|
|
1446
|
+
: `The verdict to act on is \`${effectiveVerdict}\`, NOT \`${status}\`.`;
|
|
1144
1447
|
return (`- ${id}: reported \`${status}\` — ⚠️ FLIPPED under rephrasing (${count}). ` +
|
|
1145
|
-
|
|
1448
|
+
verdictInstruction +
|
|
1146
1449
|
detail);
|
|
1147
1450
|
}
|
|
1148
1451
|
// An unreadable `flipped` is NOT "held" — say so rather than printing the
|
|
1149
1452
|
// verdict as if it had survived re-asking.
|
|
1150
|
-
if (
|
|
1151
|
-
|
|
1453
|
+
if (outcome === null) {
|
|
1454
|
+
const weakerVerdict = effectiveVerdict !== status
|
|
1455
|
+
? ` The verdict to act on is \`${effectiveVerdict}\`, NOT \`${status}\`.`
|
|
1456
|
+
: '';
|
|
1457
|
+
return `- ${id}: reported \`${status}\` — ⚠️ flip status UNREADABLE (${count}); the robustness outcome could not be read safely, so do not rely on it.${weakerVerdict}${detail}`;
|
|
1152
1458
|
}
|
|
1153
|
-
return `- ${id}: \`${
|
|
1459
|
+
return `- ${id}: \`${effectiveVerdict}\` — ${count}${detail}`;
|
|
1154
1460
|
});
|
|
1155
|
-
const
|
|
1156
|
-
|
|
1461
|
+
const outcomes = results.map((raw) => {
|
|
1462
|
+
const result = asRecord(raw);
|
|
1463
|
+
return readFinalHypothesisState(result).robustness;
|
|
1464
|
+
});
|
|
1465
|
+
const flippedCount = outcomes.filter((outcome) => outcome === 'flipped').length;
|
|
1466
|
+
const unreadableFlips = outcomes.filter((outcome) => outcome === null).length;
|
|
1157
1467
|
// The un-retested hypotheses are counted against the FULL result set, not just
|
|
1158
1468
|
// the re-tested subset: a bare "no verdict flipped" over a partially-tested run
|
|
1159
1469
|
// reads as "every verdict survived", when most may never have been re-asked.
|
|
1160
|
-
const untested =
|
|
1470
|
+
const untested = outcomes.filter((outcome) => outcome === 'not_retested').length;
|
|
1471
|
+
const retested = results.length - untested;
|
|
1161
1472
|
const untestedNote = untested > 0
|
|
1162
|
-
? ` (${
|
|
1473
|
+
? ` (${retested}/${results.length} verdicts re-tested; the other ${untested} were NOT re-asked — treat those as single-shot)`
|
|
1163
1474
|
: '';
|
|
1164
|
-
const header =
|
|
1165
|
-
?
|
|
1166
|
-
:
|
|
1167
|
-
? `### Hypothesis robustness — ⚠️
|
|
1168
|
-
:
|
|
1475
|
+
const header = untested === results.length
|
|
1476
|
+
? '### Hypothesis robustness — NOT RE-TESTED on this run'
|
|
1477
|
+
: flippedCount > 0
|
|
1478
|
+
? `### Hypothesis robustness — ⚠️ ${flippedCount} verdict(s) flipped under rephrasing${untestedNote}`
|
|
1479
|
+
: unreadableFlips > 0
|
|
1480
|
+
? `### Hypothesis robustness — ⚠️ flip status UNREADABLE for ${unreadableFlips} verdict(s) (not a clean result)${untestedNote}`
|
|
1481
|
+
: `### Hypothesis robustness — no re-tested verdict flipped${untestedNote}`;
|
|
1169
1482
|
return (`${header}\n${lines.join('\n')}` +
|
|
1170
|
-
capNote(shown.length,
|
|
1483
|
+
capNote(shown.length, renderable.length, channel, 'report_data.hypothesisResults'));
|
|
1171
1484
|
}
|
|
1172
1485
|
// ---------------------------------------------------------------------------
|
|
1173
1486
|
// Model-authored framing — prose side of the proof separator (FUL-570)
|
|
@@ -1471,7 +1784,7 @@ function renderMarketSizing(reportData) {
|
|
|
1471
1784
|
return `- ${labels[scope]}: — (Not established)`;
|
|
1472
1785
|
}
|
|
1473
1786
|
const validInputs = inputs.filter((input) => input !== null);
|
|
1474
|
-
const sameBoundary = validInputs.every((input) => input.period === entry.period && input.geography === entry.geography && input.audience === entry.audience);
|
|
1787
|
+
const sameBoundary = validInputs.every((input) => input.kind === 'ratio_assumption' || (input.period === entry.period && input.geography === entry.geography && input.audience === entry.audience));
|
|
1475
1788
|
const inputValues = validInputs.map((input) => num(input.value));
|
|
1476
1789
|
const oneCurrency = validInputs.filter((input) => input.kind === 'currency' && input.currency === currency).length === 1;
|
|
1477
1790
|
const product = inputValues.every((input) => input !== null)
|
|
@@ -1485,6 +1798,204 @@ function renderMarketSizing(reportData) {
|
|
|
1485
1798
|
});
|
|
1486
1799
|
return `### TAM / SAM / SOM — typed sizing\n${lines.join('\n')}`;
|
|
1487
1800
|
}
|
|
1801
|
+
/**
|
|
1802
|
+
* The Reddit community-evidence rail's typed outcome (FUL-685 / T8, D6).
|
|
1803
|
+
*
|
|
1804
|
+
* ⚠️ SILENCE IS A STATE, AND SO IS SAYING NOTHING — they are different, and this function is
|
|
1805
|
+
* where the difference is kept. TWO cases render NOTHING at all:
|
|
1806
|
+
*
|
|
1807
|
+
* - the note is **absent or unreadable** → UNRECORDED. Every report written before the field
|
|
1808
|
+
* existed omits it, as does one served by an app deploy that predates it. Printing "Reddit
|
|
1809
|
+
* was not searched" off that silence would be inventing a fact about the run.
|
|
1810
|
+
* - the note says **`not_run`** → the rail did not dispatch, so there is no coverage claim to
|
|
1811
|
+
* make in either direction. The plan's placement table is explicit: no Reddit-specific copy
|
|
1812
|
+
* and no claim that Reddit ran; this is operator telemetry, and it lives in the ledger.
|
|
1813
|
+
*
|
|
1814
|
+
* The other four each get their own sentence, and `failed` may NEVER be relabelled
|
|
1815
|
+
* `completed_empty` (F7): a provider outage dressed up as an honest empty search is a false claim
|
|
1816
|
+
* about coverage, which is the one thing this whole digest exists not to make.
|
|
1817
|
+
*
|
|
1818
|
+
* ⚠️ COUNTS ARE TWO NUMBERS, NOT ONE. `acceptedSourceCount` is discussions and
|
|
1819
|
+
* `acceptedExcerptCount` is excerpts, because one thread legitimately yields several — collapsing
|
|
1820
|
+
* them would let receipt multiplicity inflate the breadth figure, which D12/D13 forbid.
|
|
1821
|
+
*
|
|
1822
|
+
* ⚠️ NO COST, NO LATENCY, NO REASON CODE ON A HEALTHY RUN. `costUsd`, `latencyMs`,
|
|
1823
|
+
* `requestsIssued`, `toolRequests` and the three intent counters are operator telemetry; they
|
|
1824
|
+
* change nothing an agent should do about a claim, and a digest an agent skims past is a digest it
|
|
1825
|
+
* does not read. The reason code renders only where it explains a limitation the reader has to act
|
|
1826
|
+
* on.
|
|
1827
|
+
*
|
|
1828
|
+
* ⚠️ ONE PRODUCER WRITES THIS TODAY, AND IT IS NOT THE HERO. `redditReportMethodology`
|
|
1829
|
+
* (`agent/src/scoped-research.ts`, FUL-686 / T9) stamps the summary on every scoped run whose rail
|
|
1830
|
+
* was ARMED; the hero orchestrator still DELETES the key at its enrich chokepoint, and T10
|
|
1831
|
+
* (FUL-687) owns replacing that delete with the real backfill plus the matching terminal event. So
|
|
1832
|
+
* a report reaching this renderer with no note is either a hero run or a pre-feature one — which
|
|
1833
|
+
* is exactly why UNRECORDED, and not "Reddit was not searched", is the only honest reading of
|
|
1834
|
+
* absence.
|
|
1835
|
+
*/
|
|
1836
|
+
function renderRedditRetrieval(reportData) {
|
|
1837
|
+
const note = asRecord(asRecord(reportData.methodology)?.redditRetrieval);
|
|
1838
|
+
if (!note)
|
|
1839
|
+
return '';
|
|
1840
|
+
const outcome = str(note.outcome);
|
|
1841
|
+
if (!outcome || outcome === 'not_run')
|
|
1842
|
+
return '';
|
|
1843
|
+
const reason = safeInline(note.reason, 60);
|
|
1844
|
+
const because = reason ? ` (reason code: \`${reason}\`)` : '';
|
|
1845
|
+
const sourceCounter = readAcceptanceCounter(note.acceptedSourceCount);
|
|
1846
|
+
const excerptCounter = readAcceptanceCounter(note.acceptedExcerptCount);
|
|
1847
|
+
const sources = sourceCounter.kind === 'valid' ? sourceCounter.value : null;
|
|
1848
|
+
const excerpts = excerptCounter.kind === 'valid' ? excerptCounter.value : null;
|
|
1849
|
+
const validCounters = sourceCounter.kind === 'valid' && excerptCounter.kind === 'valid'
|
|
1850
|
+
? { sources: sourceCounter.value, excerpts: excerptCounter.value }
|
|
1851
|
+
: null;
|
|
1852
|
+
// ⚠️ THE TWO COUNTERS ARE POSITIVE TOGETHER OR ZERO TOGETHER, AND A PAIR THAT DISAGREES IS A
|
|
1853
|
+
// CORRUPT NOTE (FUL-732). Both derive from the same accepted-source array and every accepted
|
|
1854
|
+
// source carries at least one excerpt, so `acceptedSourceCount: 0` beside `acceptedExcerptCount:
|
|
1855
|
+
// 7` is not a measurement this rail can produce — it is the note telling you it drifted. Reading
|
|
1856
|
+
// acceptance off whichever number happens to be positive trusts a record that has already proved
|
|
1857
|
+
// it cannot be trusted, which is what the OR below used to do.
|
|
1858
|
+
const countersContradict = validCounters !== null && validCounters.sources > 0 !== validCounters.excerpts > 0;
|
|
1859
|
+
// Counts are stated only when BOTH are readable AND they agree. "3 discussions · unknown
|
|
1860
|
+
// excerpts" reads as a measurement rather than as a gap in the record, and the outcome sentence
|
|
1861
|
+
// already carries the fact that matters; "0 accepted discussions · 7 bounded excerpts" reads as
|
|
1862
|
+
// a measurement of something impossible, which is worse. Gating it here rather than per-arm is
|
|
1863
|
+
// what stops the impossible pair reaching any outcome's wording — including the ones below that
|
|
1864
|
+
// never ask about acceptance at all.
|
|
1865
|
+
const counts = validCounters && !countersContradict
|
|
1866
|
+
? ` ${plural(validCounters.sources, 'accepted discussion')} · ` +
|
|
1867
|
+
`${plural(validCounters.excerpts, 'bounded excerpt')}.`
|
|
1868
|
+
: '';
|
|
1869
|
+
// ⚠️ THIS QUESTION IS THREE-WAY, AND THE ORCHESTRATOR HAS COLLAPSED IT TO TWO IN BOTH
|
|
1870
|
+
// DIRECTIONS ALREADY (FUL-732). Read the history before touching these three lines:
|
|
1871
|
+
//
|
|
1872
|
+
// 1. FUL-728 keyed the unknown wording on `acceptedSourceCount` ALONE, so a note carrying
|
|
1873
|
+
// seven excerpts was reported as a pass that "may have accepted none" — a FALSE NEGATIVE.
|
|
1874
|
+
// 2. The correction made acceptance an OR over the two counters, so a note reporting `0`
|
|
1875
|
+
// sources beside `7` excerpts rendered "0 accepted discussions · 7 bounded excerpts.
|
|
1876
|
+
// Accepted evidence is real." plus the pool pointer — a line that contradicts itself,
|
|
1877
|
+
// reached by trusting whichever number was positive.
|
|
1878
|
+
//
|
|
1879
|
+
// Both are the same mistake: a two-valued test over a question with three answers. The rule that
|
|
1880
|
+
// holds is the table below, and neither arm may be folded into another.
|
|
1881
|
+
//
|
|
1882
|
+
// two valid counters agree and prove acceptance → accepted evidence is real
|
|
1883
|
+
// two valid counters agree on zero → accepted none
|
|
1884
|
+
// counters contradict, or either is absent/invalid → UNKNOWN wording, and NO pool pointer
|
|
1885
|
+
//
|
|
1886
|
+
// FUL-740 closes the repeated edge left by interpreting first and guarding second. A count is
|
|
1887
|
+
// valid only when it is a finite non-negative integer; one counter never validates its partner.
|
|
1888
|
+
// That makes the invalid state unrepresentable in either assertion below instead of relying on
|
|
1889
|
+
// every comparison to remember negatives, fractions and non-finite values separately.
|
|
1890
|
+
const acceptedProven = validCounters !== null &&
|
|
1891
|
+
!countersContradict &&
|
|
1892
|
+
validCounters.sources > 0 &&
|
|
1893
|
+
validCounters.excerpts > 0;
|
|
1894
|
+
const acceptedNoneProven = validCounters !== null &&
|
|
1895
|
+
!countersContradict &&
|
|
1896
|
+
validCounters.sources === 0 &&
|
|
1897
|
+
validCounters.excerpts === 0;
|
|
1898
|
+
// ⚠️ THE POOL POINTER BELONGS ONLY TO THE OUTCOMES THAT HAVE EVIDENCE. Appending it to all
|
|
1899
|
+
// four sent a `failed` reader looking for accepted excerpts in a pool the very next section
|
|
1900
|
+
// often declares EMPTY — a pointer to nothing, in the paragraph that just said the pass did not
|
|
1901
|
+
// complete. And even on a good run it is a POINTER, not a promise: the pool renders the sources
|
|
1902
|
+
// a GROUNDED claim resolved, so an accepted excerpt no claim cites is not there. The wording
|
|
1903
|
+
// says "any excerpt a claim cites", which is what the pool actually holds.
|
|
1904
|
+
const poolPointer = ' Any accepted excerpt a claim cites appears in the persona receipt pool below like any other ' +
|
|
1905
|
+
'cached source; they are excerpts from a discussion, not a speaker\'s exact words.';
|
|
1906
|
+
// ⚠️ `completed` MEANS AT LEAST ONE SOURCE WAS ACCEPTED — that is the runtime contract, not a
|
|
1907
|
+
// convention — so `completed` beside a zero acceptance count is INCONSISTENT INPUT, not a state.
|
|
1908
|
+
// Rendering the coverage sentence for it announces a successful pass and points the reader at a
|
|
1909
|
+
// pool the very next section may declare EMPTY: the same defect the `completed_partial` branch
|
|
1910
|
+
// below already refuses, one outcome over. The honest answer to a note that contradicts itself
|
|
1911
|
+
// is the one an unrecognised outcome gets — coverage UNKNOWN.
|
|
1912
|
+
//
|
|
1913
|
+
// ⚠️ EITHER COUNTER READING ZERO IS THAT CONTRADICTION, NOT JUST `acceptedSourceCount`
|
|
1914
|
+
// (#1063 review). This branch keyed on the source count alone, on the argument that `completed`
|
|
1915
|
+
// independently asserts a source was accepted so the outcome sentence survives an
|
|
1916
|
+
// `acceptedExcerptCount: 0`. That argument uses one side of the contradiction to validate
|
|
1917
|
+
// itself: by the same invariant the contradictory pair rests on, every accepted source carries
|
|
1918
|
+
// at least one excerpt, so `excerpts === 0` PROVES `sources === 0` — the exact shape this
|
|
1919
|
+
// branch already refuses to render as a completed pass. Keeping the outcome sentence there and
|
|
1920
|
+
// dropping only the figures announced coverage from a note already known corrupt, and pointed
|
|
1921
|
+
// at the pool while doing it. On `completed` an impossible PAIR (7 / 0) is not a separate case,
|
|
1922
|
+
// because whichever valid counter reads exactly zero already contradicts the outcome.
|
|
1923
|
+
//
|
|
1924
|
+
// A `null` is UNREADABLE, not zero, and keeps the completed wording: the note omitted or
|
|
1925
|
+
// corrupted a telemetry counter, which is not the note claiming something impossible.
|
|
1926
|
+
const completedWithNoAcceptance = outcome === 'completed' &&
|
|
1927
|
+
(sources === 0 || excerpts === 0);
|
|
1928
|
+
const body = completedWithNoAcceptance
|
|
1929
|
+
? `The Reddit pass reported \`completed\` beside a ZERO acceptance count${because}, which the ` +
|
|
1930
|
+
"rail's own contract does not allow — a completed pass accepted at least one source, and " +
|
|
1931
|
+
'every accepted source carries at least one excerpt, so neither counter can read zero. The ' +
|
|
1932
|
+
'note contradicts itself, so treat Reddit coverage as UNKNOWN and make no claim in either ' +
|
|
1933
|
+
'direction.'
|
|
1934
|
+
: outcome === 'completed'
|
|
1935
|
+
? `The Reddit pass completed.${counts}${poolPointer}`
|
|
1936
|
+
: outcome === 'completed_empty'
|
|
1937
|
+
? 'No Reddit evidence cleared this run\'s bounded search and source checks. This does NOT ' +
|
|
1938
|
+
'mean no relevant Reddit discussion exists, and it is not a finding about the market.'
|
|
1939
|
+
: outcome === 'completed_partial'
|
|
1940
|
+
? // ⚠️ A PARTIAL RUN MAY HAVE ACCEPTED NOTHING, and the three cases must not read alike.
|
|
1941
|
+
// `completed_partial` means "a cap, deadline or failure stopped the pass", which is
|
|
1942
|
+
// compatible with ZERO accepted sources — so asserting "accepted evidence is real"
|
|
1943
|
+
// unconditionally makes a claim about evidence that may not exist, and appends a
|
|
1944
|
+
// pointer to a pool the next section renders EMPTY. That is the same defect the
|
|
1945
|
+
// pool-pointer comment above describes for `failed`, one outcome over.
|
|
1946
|
+
//
|
|
1947
|
+
// ⚠️ AND AN UNPROVABLE COUNT IS ITS OWN CASE (FUL-728). This used to take the
|
|
1948
|
+
// evidence wording whenever the counter was unreadable, on the argument that a run
|
|
1949
|
+
// cut short having accepted an unknown amount is likelier to have accepted something
|
|
1950
|
+
// than nothing. That argues from likelihood, and this digest does not assert from
|
|
1951
|
+
// likelihood: the whole outcome vocabulary exists so a reader is never handed a
|
|
1952
|
+
// probable fact wearing a stated one's clothes. `completed_partial` PERMITS zero —
|
|
1953
|
+
// that is the difference from `completed` above, whose own runtime contract
|
|
1954
|
+
// guarantees at least one accepted source and so keeps its wording when the counter
|
|
1955
|
+
// is unreadable. Here nothing guarantees it.
|
|
1956
|
+
//
|
|
1957
|
+
// ⚠️ BOTH COUNTERS MUST BE VALID BEFORE EITHER CLAIM IS AVAILABLE (FUL-740). Earlier
|
|
1958
|
+
// versions let one readable counter settle the question, which repeatedly made a
|
|
1959
|
+
// malformed partner acquire meaning through `> 0` or `<= 0`. Classification now
|
|
1960
|
+
// admits only finite non-negative integers, and the pair is interpreted only after
|
|
1961
|
+
// both pass. Missing, wrong-typed, negative, fractional, NaN and Infinity therefore
|
|
1962
|
+
// share the UNKNOWN arm and can manufacture neither a figure nor a claim.
|
|
1963
|
+
//
|
|
1964
|
+
// ⚠️ AND A CONTRADICTORY PAIR IS THE FOURTH SPELLING, NOT A FOURTH RULE (FUL-732). It
|
|
1965
|
+
// lands in the SAME row of the table as the unreadable case — UNKNOWN, no counts, no
|
|
1966
|
+
// pool pointer — but it gets its own sentence, because "not recorded readably" is
|
|
1967
|
+
// false of two numbers that are both perfectly readable and simply cannot both be
|
|
1968
|
+
// true. Naming the real defect is what tells an operator to go look at the producer
|
|
1969
|
+
// instead of at a dropped telemetry field.
|
|
1970
|
+
countersContradict
|
|
1971
|
+
? `The Reddit pass was cut short${because}, so its coverage is PARTIAL. Its two ` +
|
|
1972
|
+
'acceptance counters CONTRADICT each other, which no real pass produces, so this ' +
|
|
1973
|
+
'note is corrupt on the question of what it accepted — treat Reddit as PARTIALLY ' +
|
|
1974
|
+
'SEARCHED, and make no claim about what it found.'
|
|
1975
|
+
: acceptedNoneProven
|
|
1976
|
+
? `The Reddit pass was cut short${because}, so its coverage is PARTIAL.${counts} ` +
|
|
1977
|
+
'It accepted no evidence before stopping — treat Reddit as PARTIALLY SEARCHED, ' +
|
|
1978
|
+
'not as searched and empty.'
|
|
1979
|
+
: !acceptedProven
|
|
1980
|
+
? `The Reddit pass was cut short${because}, so its coverage is PARTIAL. How much ` +
|
|
1981
|
+
'evidence it accepted before stopping was not recorded readably, and a partial ' +
|
|
1982
|
+
'pass may have accepted none — treat Reddit as PARTIALLY SEARCHED, and make no ' +
|
|
1983
|
+
'claim about what it found.'
|
|
1984
|
+
: `The Reddit pass was cut short${because}, so its coverage is PARTIAL.${counts} ` +
|
|
1985
|
+
'Accepted evidence is real; the absence of more is not evidence of absence.' +
|
|
1986
|
+
poolPointer
|
|
1987
|
+
: outcome === 'failed'
|
|
1988
|
+
? `The Reddit pass did NOT complete${because}. This run makes no claim about Reddit ` +
|
|
1989
|
+
'coverage — treat it as unsearched, NOT as searched and empty.'
|
|
1990
|
+
: // ⚠️ A LIVE BRANCH, NOT A FORMALITY — see `RedditRetrievalNoteSchema`'s `outcome`,
|
|
1991
|
+
// which is deliberately lenient so a client older than the app that answered it
|
|
1992
|
+
// reports UNKNOWN rather than falling silent. Failing closed here is the whole
|
|
1993
|
+
// reason the schema does not reject the token itself.
|
|
1994
|
+
`The Reddit pass reported an outcome this client does not recognise ` +
|
|
1995
|
+
`(\`${safeInline(outcome, 40) ?? 'unreadable'}\`)${because}. Treat Reddit coverage ` +
|
|
1996
|
+
'as UNKNOWN, and upgrade `@clien-ai/mcp` — the app is newer than this client.';
|
|
1997
|
+
return `### Reddit community evidence\n${body}`;
|
|
1998
|
+
}
|
|
1488
1999
|
/**
|
|
1489
2000
|
* Build the trust digest appended to `get_report`'s text.
|
|
1490
2001
|
*
|
|
@@ -1502,6 +2013,9 @@ export function buildTrustDigest(reportData, channel) {
|
|
|
1502
2013
|
renderReportSpine(data, channel),
|
|
1503
2014
|
renderMarketSizing(data),
|
|
1504
2015
|
renderCompetitorSourceQuality(data),
|
|
2016
|
+
// FUL-685: immediately above the pool the accepted rows land in, so a reader meets the
|
|
2017
|
+
// coverage caveat before the evidence rather than after it.
|
|
2018
|
+
renderRedditRetrieval(data),
|
|
1505
2019
|
renderReceiptPools(data, channel),
|
|
1506
2020
|
renderPersonas(data, channel),
|
|
1507
2021
|
renderSycophancy(data, channel),
|