@clien-ai/mcp 0.10.1 → 0.10.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -2
- package/dist/tools/registry.js +26 -2
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/report-digest.js +434 -13
- package/dist/tools/report-digest.js.map +1 -1
- package/dist/tools/reports.js +4 -14
- package/dist/tools/reports.js.map +1 -1
- package/dist/tools/research.js +10 -17
- package/dist/tools/research.js.map +1 -1
- package/dist/tools/status.js +4 -14
- package/dist/tools/status.js.map +1 -1
- package/dist/types/report.js +125 -0
- package/dist/types/report.js.map +1 -1
- package/package.json +1 -1
|
@@ -59,6 +59,55 @@ const CLAIM_RENDER_CAP = 40;
|
|
|
59
59
|
const CLAIM_TEXT_MAX = 140;
|
|
60
60
|
/** Max personas / hypotheses / disagreements listed before capping. */
|
|
61
61
|
const LIST_RENDER_CAP = 25;
|
|
62
|
+
/**
|
|
63
|
+
* Max chars of a hypothesis STATEMENT on a robustness row's first continuation
|
|
64
|
+
* line (FUL-614).
|
|
65
|
+
*
|
|
66
|
+
* This is the one of the three that drives the budget: `statement` is required by
|
|
67
|
+
* `HypothesisResultSchema` and was present on 371/371 hypothesis rows across the 49
|
|
68
|
+
* production jobs that carry any, so it renders on every row the section shows, and it
|
|
69
|
+
* is the only one of the three whose length nothing upstream bounds.
|
|
70
|
+
*
|
|
71
|
+
* Worst case is `LIST_RENDER_CAP` rows x (160 + ~20 for the label, indent, quotes and
|
|
72
|
+
* newline) ~= 4.5KB of statement lines on a saturated digest. ⚠️ THAT IS NOT THE WHOLE
|
|
73
|
+
* SECTION'S BUDGET: `evidenceReading` renders on nearly every row too (see
|
|
74
|
+
* {@link renderRobustness} for why its measured 3.8% is a fact about the stored corpus
|
|
75
|
+
* rather than about the field), adding 25 x (~85 for the composed sentence + ~15 of
|
|
76
|
+
* label and indent) ~= 2.5KB, so ~7KB together. A typical run has 5-8 hypotheses, so
|
|
77
|
+
* ~1.5-2KB. `scopeCaveat` stays off the budget at a real 7%.
|
|
78
|
+
*/
|
|
79
|
+
const HYPOTHESIS_STATEMENT_MAX = 160;
|
|
80
|
+
/**
|
|
81
|
+
* Max chars of a `scopeCaveat` / `evidenceReading` continuation line (FUL-614).
|
|
82
|
+
*
|
|
83
|
+
* Both are specified as ONE LINE by their producers — the synthesiser prompt asks
|
|
84
|
+
* for "a one-line note naming the mismatch", and `evidenceReading` is composed from
|
|
85
|
+
* a fixed template in `agent/src/hypothesis-reading.ts` — so this is a defensive
|
|
86
|
+
* ceiling on values that should already be well under it, in the same spirit as
|
|
87
|
+
* {@link VERDICT_SUMMARY_MAX}, not a routine trim.
|
|
88
|
+
*
|
|
89
|
+
* It can be looser than {@link HYPOTHESIS_STATEMENT_MAX} because both fields are BOUNDED
|
|
90
|
+
* BY THEIR PRODUCERS in a way `statement` is not — the longest sentence
|
|
91
|
+
* `deriveEvidenceReading`'s template can compose runs ~96 chars, and the caveat is
|
|
92
|
+
* prompted as one line — NOT because either is rare. `statement` is free model prose the
|
|
93
|
+
* synthesiser echoes from the plan under no length instruction at all, so 160 there is a
|
|
94
|
+
* real trim rather than a backstop.
|
|
95
|
+
*/
|
|
96
|
+
const HYPOTHESIS_NOTE_MAX = 200;
|
|
97
|
+
/**
|
|
98
|
+
* Max persona NAMES listed inside one group of a disagreement row (FUL-616).
|
|
99
|
+
*
|
|
100
|
+
* A separate cap from `LIST_RENDER_CAP` because it multiplies against it rather
|
|
101
|
+
* than sitting beside it: the split rows are already capped at 25, and each row
|
|
102
|
+
* now carries three name groups. Ten is comfortably above the personas a real run
|
|
103
|
+
* interviews, so this is a runaway backstop on a drifted payload and not a routine
|
|
104
|
+
* trim — the same posture `CLAIM_RENDER_CAP` takes.
|
|
105
|
+
*
|
|
106
|
+
* The overflow is STATED inline (`, +3 more`) rather than silently cut, for the
|
|
107
|
+
* reason the module header gives: a silent truncation reads as "that was all of
|
|
108
|
+
* them", and in this section that would read as "nobody else pushed back".
|
|
109
|
+
*/
|
|
110
|
+
const SPLIT_NAME_CAP = 10;
|
|
62
111
|
/**
|
|
63
112
|
* The MODEL-ATTESTED tier's user-facing word, lower-cased for inline prose (FUL-363).
|
|
64
113
|
*
|
|
@@ -109,6 +158,31 @@ const ATTESTED_WINDOW_LABEL = 'cited window (NOT a verified span)';
|
|
|
109
158
|
* carries the meaning the shared word no longer can.
|
|
110
159
|
*/
|
|
111
160
|
const UNSOURCED_LABEL = 'unsourced';
|
|
161
|
+
/**
|
|
162
|
+
* The three labels on a robustness row's continuation lines (FUL-614).
|
|
163
|
+
*
|
|
164
|
+
* ⚠️ THE LABEL IS THE GUARD, for the same reason `ATTESTED_WINDOW_LABEL`'s doc gives:
|
|
165
|
+
* `\n "…"` is EXACTLY the shape `renderReceiptPools` gives a verified persona
|
|
166
|
+
* receipt, so an unlabelled indented quote under a verdict row would read as evidence
|
|
167
|
+
* that something was checked. A hypothesis statement is the opposite — it is the
|
|
168
|
+
* assertion under test, not support for it.
|
|
169
|
+
*
|
|
170
|
+
* ⚠️ THE QUOTING IS ALSO LOAD-BEARING, and it is the prose/proof line this file's header
|
|
171
|
+
* states (FUL-510). `statement` and `scopeCaveat` are MODEL-WRITTEN, so they render as
|
|
172
|
+
* QUOTED, labelled values — attributed to the run, exactly as `renderPersonaClaim`
|
|
173
|
+
* already carries a claim's model-written `text` below the `---`. `evidenceReading` is
|
|
174
|
+
* CODE-COMPOSED (`deriveEvidenceReading` in `agent/src/hypothesis-reading.ts` is its sole
|
|
175
|
+
* writer and STRIPS any model-supplied value on the rows it does not cover), so the digest
|
|
176
|
+
* states it unquoted, in its own voice. What never crosses is a model-written JUDGEMENT
|
|
177
|
+
* about the run — that is `verdictSummary`, and it renders above the separator.
|
|
178
|
+
*
|
|
179
|
+
* `EVIDENCE_READING_LABEL` deliberately echoes the report markdown's own `- **Rests on**:`
|
|
180
|
+
* (`agent/src/report-generator.ts`). One field, one word for it, on both surfaces — the
|
|
181
|
+
* lesson `ATTESTED_LABEL` above records the hard way.
|
|
182
|
+
*/
|
|
183
|
+
const HYPOTHESIS_LABEL = 'hypothesis';
|
|
184
|
+
const SCOPE_CAVEAT_LABEL = '⚠️ scope caveat — persona support from OUTSIDE its credibility domain; directional, NOT grounded';
|
|
185
|
+
const EVIDENCE_READING_LABEL = 'rests on';
|
|
112
186
|
// ---------------------------------------------------------------------------
|
|
113
187
|
// Defensive accessors — every one of these answers "or nothing" rather than throwing
|
|
114
188
|
// ---------------------------------------------------------------------------
|
|
@@ -667,6 +741,24 @@ function renderPersonas(reportData, channel) {
|
|
|
667
741
|
* The anti-sycophancy readout — deterministically computed from the interviews,
|
|
668
742
|
* not model-opined, which is exactly why it belongs in the text: it is the one
|
|
669
743
|
* signal that says whether the personas actually pushed back or just agreed.
|
|
744
|
+
*
|
|
745
|
+
* FUL-616 gave the section its own evidence. It used to render six of the sixteen
|
|
746
|
+
* fields its four schemas declare, and the ten it withheld were the substance behind
|
|
747
|
+
* the six it printed: a headline claim about the rejection COUNT with no rejection
|
|
748
|
+
* behind it, and a split rendered as three integers when WHICH persona pushed back is
|
|
749
|
+
* usually the whole finding — "the buyer said no, the user said yes" is a different
|
|
750
|
+
* result from "1 of 3 disagreed". Names and rejections now render; the discrimination
|
|
751
|
+
* ratio and the tradeoff rows deliberately do not (see the allowlist reasons in
|
|
752
|
+
* `render-coverage.test.ts` — tradeoffs are the largest of the four shapes and the
|
|
753
|
+
* furthest from the question this section answers, so they stay the first thing to cut).
|
|
754
|
+
*
|
|
755
|
+
* ⚠️ THE COUNTS STAY BESIDE THE NAMES ON A SPLIT ROW, and that is a trust decision
|
|
756
|
+
* rather than a formatting one. A count is `array.length` — nothing a value can forge.
|
|
757
|
+
* A name is untrusted text: personas are user- or model-named, so a persona called
|
|
758
|
+
* `Ada, Marcus` renders as two supporters inside its group's parentheses. Keeping the
|
|
759
|
+
* integer in front means such a row disagrees with itself visibly (`1 supportive (Ada,
|
|
760
|
+
* Marcus)`) instead of quietly overstating who agreed with the idea — in the one
|
|
761
|
+
* section a reader consults precisely to find out whether the agreement was real.
|
|
670
762
|
*/
|
|
671
763
|
function renderSycophancy(reportData, channel) {
|
|
672
764
|
const signals = asRecord(reportData.sycophancySignals);
|
|
@@ -681,6 +773,12 @@ function renderSycophancy(reportData, channel) {
|
|
|
681
773
|
const disagreements = asArray(signals.disagreements);
|
|
682
774
|
const lines = [];
|
|
683
775
|
// A healthy run has ≥1 explicit rejection; zero is itself the warning.
|
|
776
|
+
//
|
|
777
|
+
// ⚠️ THIS LINE IS THE SECTION'S HEADLINE AND IT IS UNCHANGED BY FUL-616. The
|
|
778
|
+
// zero case is the case the section exists for — a run where nobody said no is
|
|
779
|
+
// the classic sycophancy signature — and it renders with no evidence block below
|
|
780
|
+
// it because there is no evidence, which is the finding. A change to this readout
|
|
781
|
+
// that only looks right on a run WITH rejections has broken exactly that case.
|
|
684
782
|
if (totalRejections !== null) {
|
|
685
783
|
lines.push(totalRejections === 0
|
|
686
784
|
? '⚠️ 0 explicit rejections across all personas — nobody said no to anything. A run with no ' +
|
|
@@ -693,10 +791,13 @@ function renderSycophancy(reportData, channel) {
|
|
|
693
791
|
// FUL-253: joined into one warning line, so a newline in a name would end
|
|
694
792
|
// the warning early and leave the rest of the flagged personas rendering
|
|
695
793
|
// as ordinary text below it.
|
|
696
|
-
.map((raw) =>
|
|
794
|
+
.map((raw) => personaLabel(raw));
|
|
697
795
|
lines.push(`⚠️ ${lowDiscriminationCount} persona(s) flagged \`lowDiscrimination\` (uniformly supportive AND ` +
|
|
698
796
|
`rejected nothing — their answers carry little signal): ${names.join(', ')}`);
|
|
699
797
|
}
|
|
798
|
+
const rejections = renderRejections(personaSignals, channel);
|
|
799
|
+
if (rejections)
|
|
800
|
+
lines.push(rejections);
|
|
700
801
|
if (disagreements.length > 0) {
|
|
701
802
|
const shown = disagreements.slice(0, LIST_RENDER_CAP);
|
|
702
803
|
lines.push(`Hypotheses the personas SPLIT on (disagreement is signal, not noise):\n` +
|
|
@@ -704,10 +805,9 @@ function renderSycophancy(reportData, channel) {
|
|
|
704
805
|
.map((raw) => {
|
|
705
806
|
const d = asRecord(raw);
|
|
706
807
|
const id = renderId(d?.hypothesisId);
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
return ` - ${id}: ${supportive} supportive / ${neutral} neutral / ${negative} pushed back`;
|
|
808
|
+
return (` - ${id}: ${splitGroup(d?.supportive, 'supportive')}` +
|
|
809
|
+
` / ${splitGroup(d?.neutral, 'neutral')}` +
|
|
810
|
+
` / ${splitGroup(d?.negative, 'pushed back')}`);
|
|
711
811
|
})
|
|
712
812
|
.join('\n') +
|
|
713
813
|
capNote(shown.length, disagreements.length, channel, 'report_data.sycophancySignals.disagreements'));
|
|
@@ -717,6 +817,149 @@ function renderSycophancy(reportData, channel) {
|
|
|
717
817
|
}
|
|
718
818
|
return `### Anti-sycophancy\n${lines.join('\n')}`;
|
|
719
819
|
}
|
|
820
|
+
/**
|
|
821
|
+
* The continuation lines under one robustness row (FUL-614).
|
|
822
|
+
*
|
|
823
|
+
* Returns '' when the row has nothing to add. A field that is absent, empty or
|
|
824
|
+
* whitespace-only emits NO LINE — never a label with nothing after it, which would
|
|
825
|
+
* read as "the run recorded a scope caveat and we cannot show it" on the 93% of rows
|
|
826
|
+
* that simply have none. That is `safeInline`'s own `null` contract, and here it is
|
|
827
|
+
* the whole difference between a qualification and a false alarm.
|
|
828
|
+
*
|
|
829
|
+
* ⚠️ EVERY VALUE GOES THROUGH `safeInline`, and the reason is the one `renderRobustness`
|
|
830
|
+
* already states for `status` / `downgradedStatus`: these rows are `- ` entries whose
|
|
831
|
+
* point is the ⚠️ FLIPPED warning, and `statement` / `scopeCaveat` are MODEL-WRITTEN —
|
|
832
|
+
* the first untrusted prose to reach this section. A newline in either ends the line
|
|
833
|
+
* early and lets the remainder pose as a sibling row for the same hypothesis WITHOUT the
|
|
834
|
+
* warning, which is precisely the reading a forgery would want to drop (FUL-253).
|
|
835
|
+
* `render-forgery.test.ts` has planted poisoned values on these three fields since
|
|
836
|
+
* FUL-263 — before anything rendered them, which is that file's whole premise — so its
|
|
837
|
+
* assertions bit on this change with no fixture edit.
|
|
838
|
+
*/
|
|
839
|
+
function hypothesisDetail(result) {
|
|
840
|
+
const statement = safeInline(result?.statement, HYPOTHESIS_STATEMENT_MAX);
|
|
841
|
+
const scopeCaveat = safeInline(result?.scopeCaveat, HYPOTHESIS_NOTE_MAX);
|
|
842
|
+
const evidenceReading = safeInline(result?.evidenceReading, HYPOTHESIS_NOTE_MAX);
|
|
843
|
+
return ((statement ? `\n ${HYPOTHESIS_LABEL}: "${statement}"` : '') +
|
|
844
|
+
(scopeCaveat ? `\n ${SCOPE_CAVEAT_LABEL}: "${scopeCaveat}"` : '') +
|
|
845
|
+
(evidenceReading ? `\n ${EVIDENCE_READING_LABEL}: ${evidenceReading}` : ''));
|
|
846
|
+
}
|
|
847
|
+
/**
|
|
848
|
+
* What a persona name that arrived unreadable renders as (FUL-616).
|
|
849
|
+
*
|
|
850
|
+
* One bare word carrying no punctuation of its own, so each caller decides the
|
|
851
|
+
* shape around it: `personaLabel` parenthesises it because it stands in for a
|
|
852
|
+
* whole label, and `splitGroup` prints it bare because the group already supplies
|
|
853
|
+
* the parentheses. Parenthesising it in both places produced
|
|
854
|
+
* `1 pushed back ((unnamed))` — doubled brackets that read as a formatting bug on
|
|
855
|
+
* exactly the drift path where a reader has least evidence to tell a bug from the
|
|
856
|
+
* data it is describing.
|
|
857
|
+
*/
|
|
858
|
+
const UNNAMED_PERSONA = 'unnamed';
|
|
859
|
+
/**
|
|
860
|
+
* How a persona is named wherever the anti-sycophancy section names one (FUL-616).
|
|
861
|
+
*
|
|
862
|
+
* The role is here because the name alone was ambiguous in the only place this
|
|
863
|
+
* section ever printed one — the flagged list — and FUL-616 adds a second place
|
|
864
|
+
* (the rejection rows), which would have reproduced the same ambiguity in new
|
|
865
|
+
* text. Two personas can easily share a first name in one run; "Dana (VP
|
|
866
|
+
* Engineering)" and "Dana (Founder)" are two different findings about who
|
|
867
|
+
* rubber-stamped the idea, and telling them apart is the point of naming anyone.
|
|
868
|
+
*
|
|
869
|
+
* Both halves go through `safeInline` at the budget `renderPersonaSpine` already
|
|
870
|
+
* spends on the same two fields, so the same persona is clipped identically
|
|
871
|
+
* wherever the digest names it.
|
|
872
|
+
*/
|
|
873
|
+
function personaLabel(raw) {
|
|
874
|
+
const persona = asRecord(raw);
|
|
875
|
+
const name = safeInline(persona?.personaName, 100) ?? `(${UNNAMED_PERSONA})`;
|
|
876
|
+
const role = safeInline(persona?.personaRole, 100);
|
|
877
|
+
return role ? `${name} (${role})` : name;
|
|
878
|
+
}
|
|
879
|
+
/**
|
|
880
|
+
* One stance group of a disagreement row: the count, then who (FUL-616).
|
|
881
|
+
*
|
|
882
|
+
* COUNT FIRST, deliberately — see the forgery note on `renderSycophancy`. The
|
|
883
|
+
* integer is derived from the array and cannot be influenced by what is in it, so
|
|
884
|
+
* it stays the authoritative half of the group and the names are the detail it
|
|
885
|
+
* carries. An empty group renders as the bare count (`0 neutral`) rather than as
|
|
886
|
+
* an empty pair of parentheses, matching `safeInline`'s rule that "nothing to say"
|
|
887
|
+
* gets a caller-chosen wording instead of empty punctuation.
|
|
888
|
+
*
|
|
889
|
+
* COST — this is the one cap here that MULTIPLIES, so its ceiling is stated rather
|
|
890
|
+
* than left to be derived. Worst case is `LIST_RENDER_CAP` split rows × 3 groups ×
|
|
891
|
+
* `SPLIT_NAME_CAP` names × the 100-char `safeInline` budget ≈ 76 KB, an order above
|
|
892
|
+
* anything else the digest renders. That number is a DRIFTED-PAYLOAD CEILING, not a
|
|
893
|
+
* run cost, and the two bounds it multiplies are not independent in a real run: a
|
|
894
|
+
* disagreement row exists per hypothesis and its groups partition the persona pool,
|
|
895
|
+
* so five hypotheses across five personas is 5 × 3 groups × ≤5 short names. MEASURED
|
|
896
|
+
* on the 5-persona, 3-split, 4-rejection run {@link renderRejections} also cites: all
|
|
897
|
+
* three split rows together are ~330 bytes of the section's ~1,100 — every persona in
|
|
898
|
+
* every row, the densest a run of that size gets. Even all 25 rows at that density is
|
|
899
|
+
* under 3 KB; reaching a tenth of the ceiling takes a `report_data` no
|
|
900
|
+
* synthesis run can produce — which is what a backstop is for, the posture
|
|
901
|
+
* `CLAIM_RENDER_CAP` already takes.
|
|
902
|
+
*/
|
|
903
|
+
function splitGroup(raw, label) {
|
|
904
|
+
const names = asArray(raw).map((n) => safeInline(n, 100) ?? UNNAMED_PERSONA);
|
|
905
|
+
const shown = names.slice(0, SPLIT_NAME_CAP);
|
|
906
|
+
const dropped = names.length - shown.length;
|
|
907
|
+
const who = shown.length > 0 ? ` (${shown.join(', ')}${dropped > 0 ? `, +${dropped} more` : ''})` : '';
|
|
908
|
+
return `${names.length} ${label}${who}`;
|
|
909
|
+
}
|
|
910
|
+
/**
|
|
911
|
+
* The rejection rows — the evidence behind the section's own headline (FUL-616).
|
|
912
|
+
*
|
|
913
|
+
* ⚠️ THE INTRO LINE SAYS WHAT THIS TEXT IS NOT, and that is a guard rather than a
|
|
914
|
+
* caveat. These reasons are indented quoted prose, which is the exact shape
|
|
915
|
+
* `renderReceiptPools` gives a verified persona receipt and the shape FUL-615 had to
|
|
916
|
+
* put a label in front of for the same reason. A synthetic persona's reason for
|
|
917
|
+
* saying no is not a source quote and must never be cited as one; the section's own
|
|
918
|
+
* value depends on that line staying honest, because a reader who mistook these for
|
|
919
|
+
* retrieved quotes would be trusting the sycophancy readout MORE than the report it
|
|
920
|
+
* is a check on.
|
|
921
|
+
*
|
|
922
|
+
* ⚠️ ROWS ARE NOT FILTERED FOR SUBSTANCE, on purpose. `computeSycophancySignals`
|
|
923
|
+
* already drops hollow `{ option: '', reason: '' }` rejections before storing — that
|
|
924
|
+
* filter is what makes `totalRejections` a real count rather than a presence tally —
|
|
925
|
+
* so a row reaching here with nothing in it is DRIFT, and the honest render is a row
|
|
926
|
+
* saying so. Dropping it silently would make this block disagree with the headline
|
|
927
|
+
* count above it and give a reader no way to see which number was wrong.
|
|
928
|
+
*
|
|
929
|
+
* Clipping reuses this file's existing budgets rather than minting one: `CLAIM_TEXT_MAX`
|
|
930
|
+
* for the reason (the cap the digest already spends on model-written claim prose, which
|
|
931
|
+
* is what this is) and the 100 `renderPersonaSpine` spends on a short persona field for
|
|
932
|
+
* the option label.
|
|
933
|
+
*
|
|
934
|
+
* DIGEST COST, in the idiom `RECEIPT_QUOTE_MAX`'s doc sets, because a digest an agent
|
|
935
|
+
* truncates is a digest it does not read. Bounded by construction at `LIST_RENDER_CAP`
|
|
936
|
+
* (25) rows × (140 + 100 + ~40 for the label, indent and separators) ≈ 7 KB worst case.
|
|
937
|
+
* MEASURED on a 5-persona, 4-rejection, 3-split run: the whole section is ~1,100 bytes,
|
|
938
|
+
* of which this block is ~320 — against the +8,360 FUL-247 spent on the receipt quotes.
|
|
939
|
+
* A run with ZERO rejections — the case this section exists for — adds nothing at all.
|
|
940
|
+
*/
|
|
941
|
+
function renderRejections(personaSignals, channel) {
|
|
942
|
+
const rows = [];
|
|
943
|
+
let total = 0;
|
|
944
|
+
for (const raw of personaSignals) {
|
|
945
|
+
const label = personaLabel(raw);
|
|
946
|
+
for (const rejectionRaw of asArray(asRecord(raw)?.rejections)) {
|
|
947
|
+
total += 1;
|
|
948
|
+
if (rows.length >= LIST_RENDER_CAP)
|
|
949
|
+
continue;
|
|
950
|
+
const rejection = asRecord(rejectionRaw);
|
|
951
|
+
const option = safeInline(rejection?.option, 100) ?? '(option not recorded)';
|
|
952
|
+
const reason = safeInline(rejection?.reason, CLAIM_TEXT_MAX);
|
|
953
|
+
rows.push(` - ${label} ✕ ${option}${reason ? ` — "${reason}"` : ''}`);
|
|
954
|
+
}
|
|
955
|
+
}
|
|
956
|
+
if (rows.length === 0)
|
|
957
|
+
return '';
|
|
958
|
+
return ('What the personas explicitly rejected (the evidence behind the count above — the wording is a ' +
|
|
959
|
+
'SYNTHETIC persona\'s own, never a retrieved source quote, and must not be cited as one):\n' +
|
|
960
|
+
rows.join('\n') +
|
|
961
|
+
capNote(rows.length, total, channel, 'report_data.sycophancySignals.personaSignals'));
|
|
962
|
+
}
|
|
720
963
|
/**
|
|
721
964
|
* Hypothesis verdict robustness.
|
|
722
965
|
*
|
|
@@ -724,6 +967,37 @@ function renderSycophancy(reportData, channel) {
|
|
|
724
967
|
* report's top-level `status` is NOT the verdict to act on — `downgradedStatus`
|
|
725
968
|
* is. That inversion is precisely the kind of thing an agent gets wrong when it
|
|
726
969
|
* reads only the prose, so it is stated per-hypothesis rather than as a footnote.
|
|
970
|
+
*
|
|
971
|
+
* FUL-614: a row used to identify its hypothesis BY ID ALONE, so the sharpest line in
|
|
972
|
+
* this digest named something a reader could only resolve by grepping the prose above
|
|
973
|
+
* for `H3`. It now carries the statement, and — where the run set them — the FUL-102
|
|
974
|
+
* scope caveat and the FUL-396 evidence reading, on labelled continuation lines. See
|
|
975
|
+
* {@link hypothesisDetail}.
|
|
976
|
+
*
|
|
977
|
+
* ⚠️ WHAT MAKES A FIELD THE ACTUAL GAP IS HOW OFTEN IT IS THERE — and a coverage rate
|
|
978
|
+
* says that only once it is DATED AGAINST THE PRODUCER THAT WRITES IT. Measured
|
|
979
|
+
* 2026-08-24 over the 49 stored production jobs carrying `hypothesisResults` (371 rows):
|
|
980
|
+
* `statement` 371/371, `scopeCaveat` 7%, `evidenceReading` 14/371 (3.8%). The first two
|
|
981
|
+
* are properties of the fields. The third is NOT. `withEvidenceReadings` — its sole
|
|
982
|
+
* writer, and the reason the line is code-composed rather than opined — landed
|
|
983
|
+
* 2026-08-17 (FUL-396), one week before that measurement, and
|
|
984
|
+
* `agent/src/hypothesis-reading.ts` documents 357/357 production hypotheses carrying at
|
|
985
|
+
* least one evidence item. So 3.8% is the age of the stored corpus, not the rarity of
|
|
986
|
+
* the field: it fires on essentially every row written since, and the 14 rows that carry
|
|
987
|
+
* it are consistent with being exactly that post-landing slice. `scopeCaveat`'s 7%
|
|
988
|
+
* survives the same check — its prompt has shipped since 2026-07-12 (FUL-102).
|
|
989
|
+
*
|
|
990
|
+
* So `statement` is what makes every row self-describing, `evidenceReading` joins it on
|
|
991
|
+
* nearly every row from here on, and `scopeCaveat` is the one field of the three that
|
|
992
|
+
* reaches NO other agent-visible text at all:
|
|
993
|
+
* `generateReportMarkdown` prints the statement as a heading, the evidence reading as
|
|
994
|
+
* `- **Rests on**:` and every robustness variant as its own bullet, but it has never
|
|
995
|
+
* printed the scope caveat. A verdict whose persona support came from outside that
|
|
996
|
+
* persona's credibility domain reached an agent with nothing marking it.
|
|
997
|
+
*
|
|
998
|
+
* The variants block and the evidence arrays stay out, decided with Alex on the same
|
|
999
|
+
* measurement: the markdown above the separator already prints every re-ask and every
|
|
1000
|
+
* evidence summary, and they are the largest per-row cost this section could take.
|
|
727
1001
|
*/
|
|
728
1002
|
function renderRobustness(reportData, channel) {
|
|
729
1003
|
const results = asArray(reportData.hypothesisResults);
|
|
@@ -750,16 +1024,21 @@ function renderRobustness(reportData, channel) {
|
|
|
750
1024
|
const flipped = tribool(robustness?.flipped);
|
|
751
1025
|
const downgraded = safeInline(robustness?.downgradedStatus, 40);
|
|
752
1026
|
const count = survived !== null && total !== null ? `held ${survived}/${total} framings` : 'survival count unavailable';
|
|
1027
|
+
// Appended to EVERY branch, not just the flipped one. A held verdict is the row a
|
|
1028
|
+
// reader is most likely to act on unexamined, so it is the last one that should be
|
|
1029
|
+
// identified by an id alone.
|
|
1030
|
+
const detail = hypothesisDetail(result);
|
|
753
1031
|
if (flipped === true) {
|
|
754
1032
|
return (`- ${id}: reported \`${status}\` — ⚠️ FLIPPED under rephrasing (${count}). ` +
|
|
755
|
-
`The verdict to act on is \`${downgraded ?? 'weaker than reported — downgrade unavailable'}\`, NOT \`${status}\`.`
|
|
1033
|
+
`The verdict to act on is \`${downgraded ?? 'weaker than reported — downgrade unavailable'}\`, NOT \`${status}\`.` +
|
|
1034
|
+
detail);
|
|
756
1035
|
}
|
|
757
1036
|
// An unreadable `flipped` is NOT "held" — say so rather than printing the
|
|
758
1037
|
// verdict as if it had survived re-asking.
|
|
759
1038
|
if (flipped === null) {
|
|
760
|
-
return `- ${id}: reported \`${status}\` — ⚠️ flip status UNREADABLE (${count}); treat this verdict as un-retested
|
|
1039
|
+
return `- ${id}: reported \`${status}\` — ⚠️ flip status UNREADABLE (${count}); treat this verdict as un-retested.${detail}`;
|
|
761
1040
|
}
|
|
762
|
-
return `- ${id}: \`${status}\` — ${count}`;
|
|
1041
|
+
return `- ${id}: \`${status}\` — ${count}${detail}`;
|
|
763
1042
|
});
|
|
764
1043
|
const flippedCount = withRobustness.filter((raw) => bool(asRecord(asRecord(raw)?.robustness)?.flipped)).length;
|
|
765
1044
|
const unreadableFlips = withRobustness.filter((raw) => tribool(asRecord(asRecord(raw)?.robustness)?.flipped) === null).length;
|
|
@@ -778,6 +1057,87 @@ function renderRobustness(reportData, channel) {
|
|
|
778
1057
|
return (`${header}\n${lines.join('\n')}` +
|
|
779
1058
|
capNote(shown.length, withRobustness.length, channel, 'report_data.hypothesisResults'));
|
|
780
1059
|
}
|
|
1060
|
+
// ---------------------------------------------------------------------------
|
|
1061
|
+
// Model-authored framing — prose side of the proof separator (FUL-570)
|
|
1062
|
+
// ---------------------------------------------------------------------------
|
|
1063
|
+
/** Defensive ceiling for one model-authored framing value read from raw report_data. */
|
|
1064
|
+
const MODEL_FRAMING_TEXT_MAX = 400;
|
|
1065
|
+
/** Runaway backstop for assumptions / risks; normal briefs carry only a handful. */
|
|
1066
|
+
const MODEL_FRAMING_LIST_CAP = 10;
|
|
1067
|
+
export const MODEL_FRAMING_HEADING = '## Model-authored report framing — not evidence or findings';
|
|
1068
|
+
function framingList(raw) {
|
|
1069
|
+
return asArray(raw)
|
|
1070
|
+
.map((value) => safeInline(value, MODEL_FRAMING_TEXT_MAX))
|
|
1071
|
+
.filter((value) => Boolean(value));
|
|
1072
|
+
}
|
|
1073
|
+
function renderPersonasSynthesis(reportData) {
|
|
1074
|
+
const synthesis = asRecord(reportData.personasSynthesis);
|
|
1075
|
+
if (!synthesis)
|
|
1076
|
+
return '';
|
|
1077
|
+
const headline = safeInline(synthesis.headline, MODEL_FRAMING_TEXT_MAX);
|
|
1078
|
+
const body = safeInline(synthesis.body, MODEL_FRAMING_TEXT_MAX);
|
|
1079
|
+
// The producer treats the pair atomically: a headline without its justification is not a
|
|
1080
|
+
// degraded synthesis, it is an assertion with the reasoning removed. Preserve that rule on
|
|
1081
|
+
// the raw-passthrough path too.
|
|
1082
|
+
if (!headline || !body)
|
|
1083
|
+
return '';
|
|
1084
|
+
return ('### Persona-roster synthesis — model-authored, not evidence\n\n' +
|
|
1085
|
+
`**${headline}**\n\n${body}`);
|
|
1086
|
+
}
|
|
1087
|
+
function renderResearchBrief(reportData) {
|
|
1088
|
+
const brief = asRecord(reportData.researchBrief);
|
|
1089
|
+
if (!brief)
|
|
1090
|
+
return '';
|
|
1091
|
+
const lines = [];
|
|
1092
|
+
const targetMarket = safeInline(brief.targetMarket, MODEL_FRAMING_TEXT_MAX);
|
|
1093
|
+
const valueProposition = safeInline(brief.valueProposition, MODEL_FRAMING_TEXT_MAX);
|
|
1094
|
+
const competitiveLandscape = safeInline(brief.competitiveLandscape, MODEL_FRAMING_TEXT_MAX);
|
|
1095
|
+
if (targetMarket)
|
|
1096
|
+
lines.push(`- Target market: ${targetMarket}`);
|
|
1097
|
+
if (valueProposition)
|
|
1098
|
+
lines.push(`- Value proposition: ${valueProposition}`);
|
|
1099
|
+
const assumptions = framingList(brief.assumptions);
|
|
1100
|
+
if (assumptions.length > 0) {
|
|
1101
|
+
lines.push('- Assumptions to test:');
|
|
1102
|
+
lines.push(...assumptions
|
|
1103
|
+
.slice(0, MODEL_FRAMING_LIST_CAP)
|
|
1104
|
+
.map((assumption) => ` - ${assumption}`));
|
|
1105
|
+
if (assumptions.length > MODEL_FRAMING_LIST_CAP) {
|
|
1106
|
+
lines.push(` - … ${assumptions.length - MODEL_FRAMING_LIST_CAP} more in report_data`);
|
|
1107
|
+
}
|
|
1108
|
+
}
|
|
1109
|
+
if (competitiveLandscape) {
|
|
1110
|
+
lines.push(`- Competitive landscape: ${competitiveLandscape}`);
|
|
1111
|
+
}
|
|
1112
|
+
const risks = framingList(brief.keyRisks);
|
|
1113
|
+
if (risks.length > 0) {
|
|
1114
|
+
lines.push('- Key risks:');
|
|
1115
|
+
lines.push(...risks.slice(0, MODEL_FRAMING_LIST_CAP).map((risk) => ` - ${risk}`));
|
|
1116
|
+
if (risks.length > MODEL_FRAMING_LIST_CAP) {
|
|
1117
|
+
lines.push(` - … ${risks.length - MODEL_FRAMING_LIST_CAP} more in report_data`);
|
|
1118
|
+
}
|
|
1119
|
+
}
|
|
1120
|
+
if (lines.length === 0)
|
|
1121
|
+
return '';
|
|
1122
|
+
return '### Research brief — model-generated framing, not findings\n\n' + lines.join('\n');
|
|
1123
|
+
}
|
|
1124
|
+
/**
|
|
1125
|
+
* The approved model-authored context block.
|
|
1126
|
+
*
|
|
1127
|
+
* `productName` and `problemStatement` remain typed but are deliberately not repeated: the
|
|
1128
|
+
* former is already the report title and the latter restates the caller's idea. Only the five
|
|
1129
|
+
* actionable framing fields approved in FUL-570 render here. This block is composed ABOVE the
|
|
1130
|
+
* proof separator and labels itself twice; no line can be mistaken for a graded finding.
|
|
1131
|
+
*/
|
|
1132
|
+
function renderModelFraming(reportData) {
|
|
1133
|
+
const data = asRecord(reportData);
|
|
1134
|
+
if (!data)
|
|
1135
|
+
return '';
|
|
1136
|
+
const sections = [renderPersonasSynthesis(data), renderResearchBrief(data)].filter(Boolean);
|
|
1137
|
+
if (sections.length === 0)
|
|
1138
|
+
return '';
|
|
1139
|
+
return `${MODEL_FRAMING_HEADING}\n\n${sections.join('\n\n')}`;
|
|
1140
|
+
}
|
|
781
1141
|
/**
|
|
782
1142
|
* The rendered budget for the run-level verdict rationale.
|
|
783
1143
|
*
|
|
@@ -861,6 +1221,64 @@ const NO_TRUST_DATA = `${DIGEST_HEADING}\n\n` +
|
|
|
861
1221
|
'verified against a source: there is no claim registry, no receipts, no QA flags, and no ' +
|
|
862
1222
|
'anti-sycophancy readout to check it against. Treat every figure, quote, and verdict above as ' +
|
|
863
1223
|
'UNVERIFIED — do not cite anything from it as source-checked.';
|
|
1224
|
+
export const COMPETITOR_SOURCE_QUALITY_HEADING = '### Competitor source quality — code-derived flags';
|
|
1225
|
+
const MARKDOWN_LINK_NAME = /^\[([^\]\n]+)\]\([^)\n]+\)$/;
|
|
1226
|
+
const PARTIAL_MARKDOWN_LINK_NAME = /\[[^\]\n]*\]\(/;
|
|
1227
|
+
const URL_LIKE_NAME = /(?:\bhttps?:\/\/|\bwww\.)/i;
|
|
1228
|
+
/**
|
|
1229
|
+
* Render an identifier, never a link. A competitor name is model-authored and can
|
|
1230
|
+
* itself contain Markdown link syntax or a raw URL even when the separate `url`
|
|
1231
|
+
* field is withheld. An exact Markdown link keeps only its label; other URL-like
|
|
1232
|
+
* names fail closed to a placeholder. Remaining Markdown punctuation, including
|
|
1233
|
+
* reference-link brackets and dots that Markdown may autolink, is removed.
|
|
1234
|
+
*/
|
|
1235
|
+
function competitorIdentifier(raw) {
|
|
1236
|
+
const flattened = safeInline(raw, CLAIM_TEXT_MAX);
|
|
1237
|
+
if (!flattened)
|
|
1238
|
+
return '(unnamed competitor)';
|
|
1239
|
+
const markdownLink = MARKDOWN_LINK_NAME.exec(flattened);
|
|
1240
|
+
const candidate = markdownLink?.[1] ?? flattened;
|
|
1241
|
+
if ((!markdownLink && PARTIAL_MARKDOWN_LINK_NAME.test(flattened)) ||
|
|
1242
|
+
URL_LIKE_NAME.test(candidate)) {
|
|
1243
|
+
return '(competitor name withheld: URL-like)';
|
|
1244
|
+
}
|
|
1245
|
+
const plain = collapseWhitespace(candidate.replace(/[^\p{L}\p{N}&'’+\- ]/gu, ' '));
|
|
1246
|
+
return plain || '(unnamed competitor)';
|
|
1247
|
+
}
|
|
1248
|
+
/**
|
|
1249
|
+
* FUL-587's compact disclosure. The aggregate and the decision to mark a row are code-derived
|
|
1250
|
+
* from the exact canonical `marketing_seo` token. Only the flagged rows are named; positioning,
|
|
1251
|
+
* notes, url and threat level remain model-authored prose and never cross the proof separator.
|
|
1252
|
+
*
|
|
1253
|
+
* The competitor name is the approved row identifier, not a prose block. It is flattened and
|
|
1254
|
+
* capped before rendering, so a stored name cannot mint a digest row or heading of its own.
|
|
1255
|
+
*/
|
|
1256
|
+
function renderCompetitorSourceQuality(reportData) {
|
|
1257
|
+
const competitors = asArray(reportData.competitors);
|
|
1258
|
+
if (competitors.length === 0)
|
|
1259
|
+
return '';
|
|
1260
|
+
const readable = competitors
|
|
1261
|
+
.map(asRecord)
|
|
1262
|
+
.filter((competitor) => competitor !== null);
|
|
1263
|
+
const unreadableCount = competitors.length - readable.length;
|
|
1264
|
+
const flagged = readable.filter((competitor) => canonicalContentType(competitor.contentType) === 'marketing_seo');
|
|
1265
|
+
const shown = flagged.slice(0, LIST_RENDER_CAP);
|
|
1266
|
+
const lines = shown.map((competitor) => `- ⚠️ ${competitorIdentifier(competitor.name)} — \`marketing_seo\``);
|
|
1267
|
+
const aggregate = `${flagged.length} of ${readable.length} competitor rows are flagged \`marketing_seo\`. ` +
|
|
1268
|
+
'The mark is a source-quality classification used by synthesis filters, not a claim-level ' +
|
|
1269
|
+
'grounding verdict. Check reportClaims[] states and receipts: a competitor’s own page may ' +
|
|
1270
|
+
'ground a claim about itself only after span verification. Unmarked rows are not certified ' +
|
|
1271
|
+
'as grounding evidence by this summary.' +
|
|
1272
|
+
(unreadableCount > 0
|
|
1273
|
+
? ` ${unreadableCount} additional competitor array ${unreadableCount === 1 ? 'entry was' : 'entries were'} unreadable and could not be classified; do not treat that gap as clean.`
|
|
1274
|
+
: '');
|
|
1275
|
+
const cap = flagged.length > shown.length
|
|
1276
|
+
? `\n\n_Only ${shown.length} of ${flagged.length} flagged rows are named here; read report_data.competitors[] for the full structured list._`
|
|
1277
|
+
: '';
|
|
1278
|
+
return (`${COMPETITOR_SOURCE_QUALITY_HEADING}\n\n${aggregate}` +
|
|
1279
|
+
(lines.length > 0 ? `\n\n${lines.join('\n')}` : '') +
|
|
1280
|
+
cap);
|
|
1281
|
+
}
|
|
864
1282
|
/**
|
|
865
1283
|
* Build the trust digest appended to `get_report`'s text.
|
|
866
1284
|
*
|
|
@@ -876,6 +1294,7 @@ export function buildTrustDigest(reportData, channel) {
|
|
|
876
1294
|
const sections = [
|
|
877
1295
|
renderPersonaSpine(data, channel),
|
|
878
1296
|
renderReportSpine(data, channel),
|
|
1297
|
+
renderCompetitorSourceQuality(data),
|
|
879
1298
|
renderReceiptPools(data, channel),
|
|
880
1299
|
renderPersonas(data, channel),
|
|
881
1300
|
renderSycophancy(data, channel),
|
|
@@ -952,17 +1371,19 @@ function renderReportGaps(reportData) {
|
|
|
952
1371
|
* a field can reach a caller through one tool and not the others while every test stays
|
|
953
1372
|
* green. FUL-510's `verdictSummary` is the field that made the seam worth naming.
|
|
954
1373
|
*
|
|
955
|
-
* The order is the contract. Report prose first (the run's markdown, then
|
|
956
|
-
* verdict rationale), then `---`, then
|
|
957
|
-
* digest's authority rule is that the last
|
|
958
|
-
* and anything above the separator is prose
|
|
1374
|
+
* The order is the contract. Report prose first (the run's markdown, then the explicitly
|
|
1375
|
+
* labelled model-authored framing and its own stored verdict rationale), then `---`, then
|
|
1376
|
+
* the machine-checked digest LAST — because the digest's authority rule is that the last
|
|
1377
|
+
* such section in the message is the real one, and anything above the separator is prose
|
|
1378
|
+
* that may resemble it.
|
|
959
1379
|
*/
|
|
960
1380
|
export function composeReportText(reportMarkdown, reportData, channel) {
|
|
961
1381
|
const rationale = renderVerdictRationale(reportData);
|
|
1382
|
+
const framing = renderModelFraming(reportData);
|
|
962
1383
|
// FUL-586: the completeness banner goes FIRST — before the prose it is a caveat about.
|
|
963
1384
|
const gaps = renderReportGaps(reportData);
|
|
964
1385
|
return (`${gaps ? `${gaps}\n\n---\n\n` : ''}` +
|
|
965
|
-
`${reportMarkdown}${rationale ? `\n\n${rationale}` : ''}` +
|
|
1386
|
+
`${reportMarkdown}${framing ? `\n\n${framing}` : ''}${rationale ? `\n\n${rationale}` : ''}` +
|
|
966
1387
|
`\n\n---\n\n${buildTrustDigest(reportData, channel)}`);
|
|
967
1388
|
}
|
|
968
1389
|
//# sourceMappingURL=report-digest.js.map
|