kairos-chain 3.69.0 → 3.73.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +251 -0
- data/lib/kairos_mcp/version.rb +1 -1
- data/templates/knowledge/multi_llm_review_workflow/assets/review_dashboard.html +332 -0
- data/templates/knowledge/multi_llm_review_workflow/multi_llm_review_workflow.md +39 -2
- data/templates/knowledge/multi_llm_review_workflow/scripts/render_dashboard.rb +55 -0
- data/templates/skillsets/agent/tools/agent_step.rb +7 -1
- data/templates/skillsets/multi_llm_review/lib/multi_llm_review/consensus.rb +2 -2
- data/templates/skillsets/multi_llm_review/skillset.json +2 -2
- data/templates/skillsets/multi_llm_review/test/test_multi_llm_review.rb +10 -10
- data/templates/skillsets/multi_llm_review/test/test_mutation_survivors.rb +2 -2
- data/templates/skillsets/multi_llm_review/test/test_observer_set.rb +10 -10
- data/templates/skillsets/multi_llm_review/test/test_observer_set_seams.rb +13 -13
- data/templates/skillsets/multi_llm_review/test/test_tool_wiring.rb +24 -24
- data/templates/skillsets/multi_llm_review/tools/multi_llm_review.rb +2 -2
- data/templates/skillsets/multi_llm_review/tools/multi_llm_review_collect.rb +2 -2
- data/templates/skillsets/project_manager/plugin/SKILL.md +88 -0
- data/templates/skillsets/project_manager/plugin/hooks.json +46 -0
- data/templates/skillsets/project_manager/scripts/pm_l2_report.py +597 -0
- data/templates/skillsets/project_manager/skillset.json +35 -35
- data/templates/skillsets/project_manager/test/test_pm_l2_report.py +685 -0
- metadata +6 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 4344b2bf968bbb0b650c672522c8abdee9ae34daea4f06899e60cfaf815fa489
|
|
4
|
+
data.tar.gz: e2a99d727e1ca7d404ecdd0945c29e9964220b1ec8aaca8be44f46c55b9a5e77
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 9d47b320473eb188dcc9f57655f11cd152ce0cee6e160c714396ebcd508156a5696c0c0db8eeb7ace61956cfc5e61467507a051ce6d200552d95ebe3a7f9ebf1
|
|
7
|
+
data.tar.gz: e38d3d417efb9be7c9e0274d9aca36ba7c1fd062f7ef0ecfc204536f86c99a9bbbbf025a52ee31525caab67b94a3dba88c41fcd3ab9d49313d6a609b10a2e0f8
|
data/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,257 @@ All notable changes to the `kairos-chain` gem will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [3.73.0] - 2026-08-17
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- **Round 2 of the same review returned 0 APPROVE of 3 valid seats and thirteen
|
|
12
|
+
more blocking findings, three of them saying round 1's fixes did not hold. The
|
|
13
|
+
response was to subtract.** Two of the six guards round 1 added are removed
|
|
14
|
+
rather than repaired. The `-o` output path is deleted: guarding it by comparing
|
|
15
|
+
resolved paths failed three ways — a case-only difference on a case-insensitive
|
|
16
|
+
filesystem, a hardlink, and any read input the check did not enumerate, which
|
|
17
|
+
included every L2 context and `config/pm.yml` — and each failure destroyed the
|
|
18
|
+
memo while the run printed that nothing had been written to it. The page now
|
|
19
|
+
always goes to `<data dir>/log/pm_l2_report.html`, beside the data directory so a
|
|
20
|
+
relocated instance still finds it, and under a name the operator's own `.gitignore`
|
|
21
|
+
already knows how to handle. The carried `exclude` is
|
|
22
|
+
also removed: carrying it was itself a round-1 fix, and since an exclude term is
|
|
23
|
+
a substring of document names while an inferred term is usually the item's own
|
|
24
|
+
record name, it suppressed the item's own primary record. It still applies to
|
|
25
|
+
the authored terms it was written beside, and the row says when it was not
|
|
26
|
+
applied.
|
|
27
|
+
- **The anti-flood cap bounds the row, not only each term.** Twelve terms each
|
|
28
|
+
under the twenty-document cap unioned to 124 of 1179 documents for one item, and
|
|
29
|
+
a note of the ordinary shape reached 23. A row over the cap is refused rather
|
|
30
|
+
than truncated, because truncation is silent and moves `last_activity` and the
|
|
31
|
+
headline figures with it; the row reports how many its terms reached.
|
|
32
|
+
- **Paths are derived from the script's own location** instead of by appending a
|
|
33
|
+
literal `.kairos`, which had reported a populated instance as empty at exit 0 on
|
|
34
|
+
every session of a relocated data directory. **A row with no comparison names
|
|
35
|
+
which of five things is missing**, separating an unparseable L2 date from an
|
|
36
|
+
unparseable memo marker, because they blame different files and collapsing them
|
|
37
|
+
stated a false fact for the second time. **Nested store shapes are checked**, not
|
|
38
|
+
only their absence: a `projects` or `items` value that was truthy and not an
|
|
39
|
+
object reached `.values()` and raised.
|
|
40
|
+
- **The test file is rewritten, not extended.** An audit applied 65 one-line
|
|
41
|
+
mutations to the round-1 suite and 36 survived — the whole of `main`'s wiring,
|
|
42
|
+
all cross-item aggregation, and the impossible-date guard the suite was named
|
|
43
|
+
for, whose fixture built a second document matching the same term so the valid
|
|
44
|
+
date sorted last and the impossible one never reached the parse point. 53 cases
|
|
45
|
+
now, and of the 30 mutations that map to a reported finding 29 are killed; the
|
|
46
|
+
survivor is equivalent. Four habits are stated in the file: exercise guards
|
|
47
|
+
through `main` in a subprocess, check that a fixture cannot satisfy its own
|
|
48
|
+
assertion, build several rows when testing aggregation, and assert messages and
|
|
49
|
+
exit codes by content. Nothing runs the suite automatically — `rake test`
|
|
50
|
+
collects Ruby files only.
|
|
51
|
+
- **Accepted rather than fixed, and recorded where they happen.** A context
|
|
52
|
+
declaring a parseable but absurd date such as `9999-12-31` flattens the other lag
|
|
53
|
+
bars, which is what bars relative to the widest lag mean. And what the report
|
|
54
|
+
says still depends on which `python3` resolves, since `date.fromisoformat`
|
|
55
|
+
accepts basic format from 3.11. Separately, two stale `.pyc` files sit inside one
|
|
56
|
+
installed SkillSet from earlier runs, so its `content_hash` differs from a clean
|
|
57
|
+
tree's; nothing here removes them.
|
|
58
|
+
- Live data after the change: 28 of 28 items comparable, 19 with L2 more recent,
|
|
59
|
+
median 29 days, widest 81 — identical to before, so nothing the subtraction
|
|
60
|
+
removed was carrying coverage.
|
|
61
|
+
|
|
62
|
+
## [3.72.0] - 2026-08-17
|
|
63
|
+
|
|
64
|
+
**Built and never published.** Its contents ship in 3.73.0, which carries the second round of fixes to the same feature; the entry is kept because the commits it describes are in the history.
|
|
65
|
+
|
|
66
|
+
### Fixed
|
|
67
|
+
|
|
68
|
+
- **Six defects in the session-start report below, all found by a pre-release
|
|
69
|
+
multi-LLM review and all demonstrated by running code rather than reasoning
|
|
70
|
+
about it.** The review returned 0 APPROVE of 3 counted seats. Two of the five
|
|
71
|
+
invariants the change declared were false as stated, which is why they are
|
|
72
|
+
listed here rather than deferred.
|
|
73
|
+
- *The read-only promise was prose.* `open(out, "w")` accepted any path, so
|
|
74
|
+
`-o <memo>` truncated `store.json` while the run printed that nothing had been
|
|
75
|
+
written to it. The output path is now refused if it resolves to the memo or
|
|
76
|
+
the mapping, via `realpath`, so a symlink cannot walk around it.
|
|
77
|
+
- *A read-only report changed the SkillSet's chain-recorded hash.*
|
|
78
|
+
`exec_module` writes bytecode, and `scripts/__pycache__/l2_scan.cpython-310.pyc`
|
|
79
|
+
landed inside the SkillSet directory — inside `Skillset#all_file_hashes`, hence
|
|
80
|
+
inside `content_hash`, hence inside what `skillset_manager` records as
|
|
81
|
+
`skillset_event` and verifies with `raise SecurityError`. The recorded hash was
|
|
82
|
+
a function of the local CPython build and of whether the report had run.
|
|
83
|
+
`sys.dont_write_bytecode` is now set around the import.
|
|
84
|
+
- *The anti-flood guard was capped on one tier only.* A document's name is
|
|
85
|
+
`name:` or `title:` or its basename, and 321 of 1177 contexts take it from a
|
|
86
|
+
free-text `title:`. One context titled `Review` made `review` a search term
|
|
87
|
+
reaching 351 records; titled `Context`, all 1178. Both tiers are now capped at
|
|
88
|
+
twenty documents, which costs nothing measurable: of 1125 distinct names none
|
|
89
|
+
reaches more than 20, and the two items that use inference return the same 2
|
|
90
|
+
and 17 records.
|
|
91
|
+
- *A non-string `touched_at` or `due` crashed the report.* `pm_item` writes both
|
|
92
|
+
through with no check beyond a JSON type and this SkillSet's own Ruby suite
|
|
93
|
+
writes the integer `20260701` to each, so slicing them raised `TypeError` — the
|
|
94
|
+
defect `lib/project_manager/parsed_time.rb` exists to prevent, re-acquired at a
|
|
95
|
+
fifth call site because the reader is in another language. Permanent once
|
|
96
|
+
triggered, since the value stays in the store.
|
|
97
|
+
- *One shape-valid impossible date took the whole run down.* `l2_scan` validates
|
|
98
|
+
that a declared date looks like a date, not that it exists, so `2026-02-30`
|
|
99
|
+
reached `date.fromisoformat` outside the `try` — the guard covered one end of
|
|
100
|
+
the interval only. Both ends now go through it.
|
|
101
|
+
- *An unreadable memo marker was reported as unreadable records.* An item with
|
|
102
|
+
seventeen perfectly datable records was described as "records found, none
|
|
103
|
+
datable", and it silently left the denominator: 28/28 matched with median 29
|
|
104
|
+
became 27/28 with median 28.5. The three causes — no terms, undatable records,
|
|
105
|
+
unreadable marker — are now counted and worded apart.
|
|
106
|
+
- **Smaller fixes from the same review.** The authored `exclude` is carried into
|
|
107
|
+
the inference fallback rather than dropped, so a distinction the operator wrote
|
|
108
|
+
down is not undone by the fallback. A mapping whose `include` is a string rather
|
|
109
|
+
than a list no longer expands to single-character terms (one of which matched
|
|
110
|
+
1177 of 1177 documents, displayed as the operator's own authored mapping).
|
|
111
|
+
`store['projects']`, `mapping['items']` and dependency entries missing `kind` or
|
|
112
|
+
`ref` no longer raise. `--open` falls back to `xdg-open` and survives neither
|
|
113
|
+
binary existing, since the gem ships to Linux. The hook reads `KAIROS_DATA_DIR`
|
|
114
|
+
before `$CLAUDE_PROJECT_DIR/.kairos`, because the data directory is relocatable
|
|
115
|
+
and a relocated instance got a hook pointing at nothing. The hook no longer
|
|
116
|
+
discards stderr or forces a zero exit: doing both made every failure
|
|
117
|
+
indistinguishable from a session where nothing had drifted. Operator-facing text
|
|
118
|
+
no longer renders a Python list repr, and `--quiet` documents the two lines it
|
|
119
|
+
prints instead of three.
|
|
120
|
+
- **`test/test_pm_l2_report.py`, 30 cases, 27 of them red against the version that
|
|
121
|
+
shipped before them.** Of those 27, nine exercise the old behaviour directly and
|
|
122
|
+
eighteen fail because the guard function did not exist — that distinction is
|
|
123
|
+
recorded rather than counted as thirty demonstrations. The remaining three cover
|
|
124
|
+
HTML escaping and a degenerate render, which were already correct and are held
|
|
125
|
+
as regression guards. The suite drives the real `l2_scan.match` rather than a
|
|
126
|
+
copy of it, so the two views cannot disagree about what a term matched.
|
|
127
|
+
|
|
128
|
+
### Added
|
|
129
|
+
|
|
130
|
+
- **The `project_manager` SkillSet (v0.6.0) reports the memo-versus-L2 drift once
|
|
131
|
+
per session, and no longer needs a human to author search terms first.** The
|
|
132
|
+
memo lags the context store and the lag is invisible from whichever side is
|
|
133
|
+
being read: on the development instance, 19 of 28 items have L2 activity more
|
|
134
|
+
recent than their last memo touch, median 29 days, widest 81. Derivation could
|
|
135
|
+
already measure that, but only when someone remembered to run it and only for
|
|
136
|
+
items a human had authored terms for. Two read-only additions.
|
|
137
|
+
`scripts/pm_l2_report.py` renders the comparison as one HTML page, and
|
|
138
|
+
`plugin/hooks.json` declares it as a `SessionStart` hook so the harness runs it
|
|
139
|
+
without anyone deciding to. `SessionStart` and not `Stop`: this is not a gate,
|
|
140
|
+
it decides nothing, and only Stop-family payloads carry the once-per-turn brake
|
|
141
|
+
`kairos_hook_projector`'s gates need. Delivery is by projection — `skillset
|
|
142
|
+
install` changes no host settings, and the MCP handshake projects on every host
|
|
143
|
+
start, so no core change was needed to reach the host. Whether the newly
|
|
144
|
+
written hook fires on that same start or the following one depends on when the
|
|
145
|
+
host reads its settings relative to the handshake; that ordering is not
|
|
146
|
+
measured, and `kairos-plugin-project` run by hand settles it. `skillset upgrade
|
|
147
|
+
--apply` is not a substitute: it projects only when it actually upgrades
|
|
148
|
+
something, and `project_manager` is not in the core set it upgrades. Measured
|
|
149
|
+
at 0.18s over 1172 contexts; the script writes exactly one file, its own
|
|
150
|
+
output, and the memo's hash is unchanged across a run.
|
|
151
|
+
- **Search terms fall back to inference from the item's own title and notes**
|
|
152
|
+
when the authored mapping has no entry for an item, or its entry matched
|
|
153
|
+
nothing — so no item goes unreported and L2 is never asked to be relabelled.
|
|
154
|
+
Inference cannot replace the mapping and does not try: 43 of 53 hand-authored
|
|
155
|
+
terms appear nowhere in any item's title or notes, having been written from
|
|
156
|
+
knowledge of the work. What it can do is refuse to flood. Only a token that is
|
|
157
|
+
itself the name of an existing L2 document, or a compound identifier reaching
|
|
158
|
+
at most twenty documents, is accepted; a bare English word is refused however
|
|
159
|
+
rare it looks. Accepting bare words returned 51 and 82 records for the two
|
|
160
|
+
items that have almost none, because a defect is described with words like
|
|
161
|
+
store, write, config and yaml, and those match hundreds of unrelated names as
|
|
162
|
+
substrings. Refusing them, the same two return 2 and 17, and all 28 items
|
|
163
|
+
become comparable. The cost is misses — inference alone found 87 records across
|
|
164
|
+
the 24 mapped items where the mapping found 296 — and the trade is deliberate:
|
|
165
|
+
a miss shows up as a smaller count, a spurious record does not show up at all.
|
|
166
|
+
Unchanged by this release: nothing writes to the memo (derivation v0.12 §1),
|
|
167
|
+
and neither the digest's shape nor the secretary's grant moved.
|
|
168
|
+
|
|
169
|
+
## [3.71.0] - 2026-08-17
|
|
170
|
+
|
|
171
|
+
### Added
|
|
172
|
+
|
|
173
|
+
- **The round dashboard ships with the L1 `multi_llm_review_workflow` entry**
|
|
174
|
+
(v3.10.2). `scripts/render_dashboard.rb` reads a round summary as JSON on
|
|
175
|
+
stdin, fills `assets/review_dashboard.html`, and writes a self-contained page.
|
|
176
|
+
These are the worked example `resource_render` names in its own description
|
|
177
|
+
and in its default output derivation (`render_dashboard.rb` →
|
|
178
|
+
`dashboard.html`), and neither was in the distribution — they existed on one
|
|
179
|
+
instance only. A fresh install therefore had a core tool whose documented
|
|
180
|
+
example pointed at files that were not on disk. `dashboard.html` itself is
|
|
181
|
+
deliberately not shipped: it is one June 2026 render of
|
|
182
|
+
`kairos_hook_projector_stage1_design_v0.1`, and the renderer recreates it.
|
|
183
|
+
|
|
184
|
+
### Fixed
|
|
185
|
+
|
|
186
|
+
- **An L1 entry's `assets/` and `scripts/` are deleted by an upgrade without
|
|
187
|
+
appearing in the report.** The 3.70.0 upgrade removed three files from this
|
|
188
|
+
instance's `multi_llm_review_workflow` entry (two dashboard pages and the
|
|
189
|
+
renderer, 30.9 KB, last touched 2026-06-02) while the L1 section of the report
|
|
190
|
+
said only `[UPDATED] multi_llm_review_workflow` and `Conflicts: 0`. Mechanism:
|
|
191
|
+
`UpgradeAnalyzer#analyze_knowledge` hashes `<name>.md` alone to decide
|
|
192
|
+
new / unchanged / updated / user_modified / conflict, and the apply step
|
|
193
|
+
replaces the whole entry directory, so any subdirectory content the
|
|
194
|
+
distribution does not carry is removed silently. The L0 section, by contrast,
|
|
195
|
+
reports `[KEPT] … (user-modified)`. Shipping the two files removes the
|
|
196
|
+
deletion for this entry; the general reporting gap is unfixed and recorded.
|
|
197
|
+
Note for anyone adding assets to a shipped entry: the `.md` must change in the
|
|
198
|
+
same release, or the entry is classified `:unchanged` and the new files never
|
|
199
|
+
install.
|
|
200
|
+
- **The dashboard's gate stated a closing condition the project does not use.**
|
|
201
|
+
It required every blocking-pool seat to APPROVE *and* the displayed round's
|
|
202
|
+
entire (a)+(b) count to be zero. Findings may now carry `carryover: true`
|
|
203
|
+
(raised in an earlier round, still open; an absent flag means new), and the
|
|
204
|
+
panel reports a **freeze candidate** when new (a)+(b) is zero, showing the vote
|
|
205
|
+
tally beside it as a reference value. Driven through the real renderer and the
|
|
206
|
+
real gate function: a round with zero new and one carryover (a) at 1 of 2 seats
|
|
207
|
+
approving reports `GATE NOT PASSED` under the old rule and `FREEZE CANDIDATE`
|
|
208
|
+
under this one — and that is the state both 2026-08 review threads actually
|
|
209
|
+
closed in. A round with one new (a) and one carryover (a) at 2 of 2 approving
|
|
210
|
+
reports `NOT CLOSED — 1 new blocking P0 (a+b), 1 carryover`.
|
|
211
|
+
- `Consensus.aggregate`'s `@return` line documented a `:convergence` key the
|
|
212
|
+
method no longer returns; the 3.70.0 rename matched bracket and definition
|
|
213
|
+
forms only. Comment only.
|
|
214
|
+
|
|
215
|
+
## [3.70.0] - 2026-08-17
|
|
216
|
+
|
|
217
|
+
### Changed
|
|
218
|
+
|
|
219
|
+
- **The block of vote counts `multi_llm_review` returns is named `vote_tally`,
|
|
220
|
+
not `convergence`** (SkillSet v0.10.1). It holds `approve_count`,
|
|
221
|
+
`reject_count`, `skip_count`, `successful_count`, `threshold` and `rule` —
|
|
222
|
+
vote arithmetic. The closing condition the L1 workflow states is
|
|
223
|
+
"new (a)+(b) P0 = 0, carryover counted separately", which this SkillSet does
|
|
224
|
+
not compute and no returned field carries. INV-R2 had already demoted the
|
|
225
|
+
ratio to a recorded reference value in a comment at `consensus.rb:156` while
|
|
226
|
+
leaving the field's name intact, so every round handed the orchestrator a
|
|
227
|
+
block whose name claimed what its contents could not answer. Renamed rather
|
|
228
|
+
than given a new sibling field: with nothing in the payload called
|
|
229
|
+
convergence, the criterion has to be fetched from the findings, whereas a new
|
|
230
|
+
field would have left the misleading name in place. 62 sites across 8 files —
|
|
231
|
+
one definition, four production reads, 57 test assertions. One SkillSet
|
|
232
|
+
boundary is crossed: `agent`'s `agent_step.rb` reads this column and now reads
|
|
233
|
+
`vote_tally` with a fallback to `convergence`, following the pattern that file
|
|
234
|
+
already uses for the v0.7 `verdict` → `reference_verdict` rename, because
|
|
235
|
+
records written earlier still say `convergence`. Consumers outside this
|
|
236
|
+
repository that read the old key get nil. Falsified: with the three
|
|
237
|
+
production files reverted to the old name and the tests left renamed, the five
|
|
238
|
+
affected test files produce 44 errors; restored, the suite is 556 runs / 1809
|
|
239
|
+
assertions / 0 failures. Merged at the operator's instruction without
|
|
240
|
+
multi-LLM review.
|
|
241
|
+
- **The L1 `multi_llm_review_workflow` pre-flight checklist states the closing
|
|
242
|
+
condition, not the ratio** (v3.10.1). The checklist line read "Convergence
|
|
243
|
+
rule: 3/5 APPROVE (full) or 3/4 APPROVE (after exclusion)", while
|
|
244
|
+
§ Convergence Rules — 200 lines further down a 1578-line file — states that
|
|
245
|
+
the machine-side signal is "new (a)+(b) P0 = 0" and the ratio is auxiliary.
|
|
246
|
+
The checklist is what gets read before dispatch, so the ratio was the
|
|
247
|
+
operative rule in practice regardless of the prose. The line now leads with
|
|
248
|
+
the closing condition and keeps both ratios beside it as reference values. No
|
|
249
|
+
rule changed; the order in which a reader meets them did.
|
|
250
|
+
|
|
251
|
+
### Known issue, recorded rather than fixed
|
|
252
|
+
|
|
253
|
+
- `consensus[:vote_tally][:reason]` is read in two places
|
|
254
|
+
(`multi_llm_review.rb:548`, `multi_llm_review_collect.rb:431`) and is never
|
|
255
|
+
written, so the `|| 'quorum not met'` fallback is the only reachable value.
|
|
256
|
+
Changing it would alter an operator-facing message, which is a new claim.
|
|
257
|
+
|
|
7
258
|
## [3.69.0] - 2026-08-15
|
|
8
259
|
|
|
9
260
|
### Added
|
data/lib/kairos_mcp/version.rb
CHANGED
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
<!DOCTYPE html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="UTF-8">
|
|
5
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
|
6
|
+
<title>Multi-LLM Review Dashboard</title>
|
|
7
|
+
<style>
|
|
8
|
+
:root {
|
|
9
|
+
--bg: #0d1117; --surface: #161b22; --border: #30363d;
|
|
10
|
+
--text: #e6edf3; --text-muted: #8b949e;
|
|
11
|
+
--approve: #3fb950; --reject: #f85149; --revise: #d29922;
|
|
12
|
+
--advisory: #58a6ff; --font: -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif;
|
|
13
|
+
}
|
|
14
|
+
* { margin: 0; padding: 0; box-sizing: border-box; }
|
|
15
|
+
body { font-family: var(--font); background: var(--bg); color: var(--text); padding: 24px; }
|
|
16
|
+
h1 { font-size: 1.4rem; margin-bottom: 4px; }
|
|
17
|
+
.subtitle { color: var(--text-muted); font-size: 0.85rem; margin-bottom: 24px; }
|
|
18
|
+
.grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 16px; margin-bottom: 24px; }
|
|
19
|
+
.card { background: var(--surface); border: 1px solid var(--border); border-radius: 8px; padding: 16px; }
|
|
20
|
+
.card h2 { font-size: 0.95rem; color: var(--text-muted); margin-bottom: 12px; text-transform: uppercase; letter-spacing: 0.05em; }
|
|
21
|
+
.verdict { display: inline-block; padding: 2px 10px; border-radius: 12px; font-size: 0.8rem; font-weight: 600; }
|
|
22
|
+
.verdict.approve { background: rgba(63,185,80,0.15); color: var(--approve); }
|
|
23
|
+
.verdict.reject { background: rgba(248,81,73,0.15); color: var(--reject); }
|
|
24
|
+
.verdict.revise { background: rgba(210,153,34,0.15); color: var(--revise); }
|
|
25
|
+
table { width: 100%; border-collapse: collapse; font-size: 0.85rem; }
|
|
26
|
+
th, td { text-align: left; padding: 8px 12px; border-bottom: 1px solid var(--border); }
|
|
27
|
+
th { color: var(--text-muted); font-weight: 500; }
|
|
28
|
+
.finding-tag { display: inline-block; padding: 1px 6px; border-radius: 4px; font-size: 0.75rem; font-weight: 600; margin-right: 4px; }
|
|
29
|
+
.finding-a { background: rgba(248,81,73,0.15); color: var(--reject); }
|
|
30
|
+
.finding-b { background: rgba(210,153,34,0.15); color: var(--revise); }
|
|
31
|
+
.finding-c { background: rgba(88,166,255,0.15); color: var(--advisory); }
|
|
32
|
+
.stat-number { font-size: 2rem; font-weight: 700; }
|
|
33
|
+
.stat-label { color: var(--text-muted); font-size: 0.8rem; }
|
|
34
|
+
.bar-row { display: flex; align-items: center; gap: 8px; margin-bottom: 6px; }
|
|
35
|
+
.bar-label { width: 120px; font-size: 0.8rem; color: var(--text-muted); text-align: right; }
|
|
36
|
+
.bar-track { flex: 1; height: 20px; background: var(--border); border-radius: 4px; overflow: hidden; display: flex; }
|
|
37
|
+
.bar-seg { height: 100%; transition: width 0.3s; }
|
|
38
|
+
.bar-seg.a { background: var(--reject); }
|
|
39
|
+
.bar-seg.b { background: var(--revise); }
|
|
40
|
+
.bar-seg.c { background: var(--advisory); }
|
|
41
|
+
.round-tabs { display: flex; gap: 8px; margin-bottom: 16px; }
|
|
42
|
+
.round-tab { padding: 6px 16px; border-radius: 6px; border: 1px solid var(--border); background: transparent;
|
|
43
|
+
color: var(--text-muted); cursor: pointer; font-size: 0.85rem; }
|
|
44
|
+
.round-tab.active { background: var(--surface); color: var(--text); border-color: var(--advisory); }
|
|
45
|
+
.legend { display: flex; gap: 16px; margin-bottom: 16px; font-size: 0.8rem; color: var(--text-muted); }
|
|
46
|
+
.legend-item { display: flex; align-items: center; gap: 4px; }
|
|
47
|
+
.legend-dot { width: 10px; height: 10px; border-radius: 2px; }
|
|
48
|
+
.instructions { background: var(--surface); border: 1px solid var(--border); border-radius: 8px;
|
|
49
|
+
padding: 16px; margin-bottom: 24px; font-size: 0.85rem; color: var(--text-muted); }
|
|
50
|
+
.instructions code { background: var(--border); padding: 2px 6px; border-radius: 4px; font-size: 0.8rem; color: var(--text); }
|
|
51
|
+
textarea { width: 100%; min-height: 120px; background: var(--bg); border: 1px solid var(--border);
|
|
52
|
+
border-radius: 6px; padding: 12px; color: var(--text); font-family: monospace; font-size: 0.8rem; resize: vertical; }
|
|
53
|
+
button { padding: 8px 16px; border-radius: 6px; border: 1px solid var(--border); background: var(--surface);
|
|
54
|
+
color: var(--text); cursor: pointer; font-size: 0.85rem; }
|
|
55
|
+
button:hover { border-color: var(--advisory); }
|
|
56
|
+
button.primary { background: rgba(88,166,255,0.15); border-color: var(--advisory); color: var(--advisory); }
|
|
57
|
+
.actions { display: flex; gap: 8px; margin-top: 12px; }
|
|
58
|
+
#json-error { color: var(--reject); font-size: 0.8rem; margin-top: 4px; min-height: 1.2em; }
|
|
59
|
+
</style>
|
|
60
|
+
</head>
|
|
61
|
+
<body>
|
|
62
|
+
|
|
63
|
+
<h1>Multi-LLM Review Dashboard</h1>
|
|
64
|
+
<p class="subtitle">KairosChain L1: multi_llm_review_workflow — HTML resource (assets/)</p>
|
|
65
|
+
|
|
66
|
+
<div class="instructions">
|
|
67
|
+
<strong>How to use:</strong> Paste a review result JSON into the editor below, or use the sample data to explore.
|
|
68
|
+
The JSON format follows the <code>(a)/(b)/(c)</code> finding classification from the multi-LLM review workflow.
|
|
69
|
+
After visualizing, use <strong>Copy as Prompt</strong> to feed the analysis back into Claude Code.
|
|
70
|
+
</div>
|
|
71
|
+
|
|
72
|
+
<div class="card" style="margin-bottom: 24px;">
|
|
73
|
+
<h2>Review Data Input</h2>
|
|
74
|
+
<textarea id="json-input" placeholder='Paste review JSON here...'></textarea>
|
|
75
|
+
<div id="json-error"></div>
|
|
76
|
+
<div class="actions">
|
|
77
|
+
<button class="primary" onclick="loadData()">Visualize</button>
|
|
78
|
+
<button onclick="loadSample()">Load Sample</button>
|
|
79
|
+
<button onclick="copyAsPrompt()">Copy as Prompt</button>
|
|
80
|
+
</div>
|
|
81
|
+
</div>
|
|
82
|
+
|
|
83
|
+
<div id="dashboard" style="display:none;">
|
|
84
|
+
|
|
85
|
+
<div class="round-tabs" id="round-tabs"></div>
|
|
86
|
+
|
|
87
|
+
<div class="legend">
|
|
88
|
+
<div class="legend-item"><div class="legend-dot" style="background:var(--reject)"></div> (a) deployment-grounded</div>
|
|
89
|
+
<div class="legend-item"><div class="legend-dot" style="background:var(--revise)"></div> (b) philosophy-aligned</div>
|
|
90
|
+
<div class="legend-item"><div class="legend-dot" style="background:var(--advisory)"></div> (c) value-divergent</div>
|
|
91
|
+
</div>
|
|
92
|
+
|
|
93
|
+
<div class="grid">
|
|
94
|
+
<div class="card">
|
|
95
|
+
<h2>Consensus</h2>
|
|
96
|
+
<div id="consensus-verdicts"></div>
|
|
97
|
+
<div style="margin-top:12px;">
|
|
98
|
+
<span class="stat-label">New blocking P0 (a+b): </span>
|
|
99
|
+
<span id="blocking-count" class="stat-number" style="font-size:1.4rem;"></span>
|
|
100
|
+
<span class="stat-label" style="margin-left:16px;">Carryover (a+b): </span>
|
|
101
|
+
<span id="carryover-count" class="stat-number" style="font-size:1.4rem;"></span>
|
|
102
|
+
<span class="stat-label" style="margin-left:16px;">Advisory (c): </span>
|
|
103
|
+
<span id="advisory-count" class="stat-number" style="font-size:1.4rem; color:var(--advisory);"></span>
|
|
104
|
+
</div>
|
|
105
|
+
</div>
|
|
106
|
+
<div class="card">
|
|
107
|
+
<h2>Gate Status</h2>
|
|
108
|
+
<div id="gate-status" style="font-size:1.1rem; margin-top:8px;"></div>
|
|
109
|
+
</div>
|
|
110
|
+
</div>
|
|
111
|
+
|
|
112
|
+
<div class="card" style="margin-bottom:16px;">
|
|
113
|
+
<h2>Finding Distribution by Reviewer</h2>
|
|
114
|
+
<div id="finding-bars"></div>
|
|
115
|
+
</div>
|
|
116
|
+
|
|
117
|
+
<div class="card">
|
|
118
|
+
<h2>All Findings</h2>
|
|
119
|
+
<table>
|
|
120
|
+
<thead><tr><th>Reviewer</th><th>Class</th><th>Finding</th><th>Severity</th></tr></thead>
|
|
121
|
+
<tbody id="findings-table"></tbody>
|
|
122
|
+
</table>
|
|
123
|
+
</div>
|
|
124
|
+
</div>
|
|
125
|
+
|
|
126
|
+
<script>
|
|
127
|
+
const SAMPLE = {
|
|
128
|
+
artifact: "kairos_hook_projector_stage1_design_v0.1",
|
|
129
|
+
rounds: [
|
|
130
|
+
{
|
|
131
|
+
round: 1,
|
|
132
|
+
reviewers: [
|
|
133
|
+
{
|
|
134
|
+
id: "philosophy_persona",
|
|
135
|
+
label: "Philosophy Persona (4.7)",
|
|
136
|
+
pool: "blocking",
|
|
137
|
+
verdict: "APPROVE",
|
|
138
|
+
findings: [
|
|
139
|
+
{ class: "b", text: "Inv-C2: substrate reference (plugin_projector) should be invariant, not named dependency", severity: "P0" },
|
|
140
|
+
{ class: "c", text: "Consider adding explicit Prop 5 recording annotation", severity: "P2" }
|
|
141
|
+
]
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
id: "engineering_persona",
|
|
145
|
+
label: "Engineering Persona (4.7)",
|
|
146
|
+
pool: "blocking",
|
|
147
|
+
verdict: "REVISE",
|
|
148
|
+
findings: [
|
|
149
|
+
{ class: "a", text: "Inv-2 vs Inv-5 timestamp semantics contradict each other", severity: "P0" },
|
|
150
|
+
{ class: "a", text: "Orphaned compile record lifecycle not specified", severity: "P0" },
|
|
151
|
+
{ class: "b", text: "Inv-O1 declaration-order is mechanism dressed as invariant", severity: "P0" },
|
|
152
|
+
{ class: "c", text: "Prefer explicit error types over string matching", severity: "P2" }
|
|
153
|
+
]
|
|
154
|
+
},
|
|
155
|
+
{
|
|
156
|
+
id: "claude_cli_4.6",
|
|
157
|
+
label: "Claude CLI (Opus 4.6)",
|
|
158
|
+
pool: "blocking",
|
|
159
|
+
verdict: "REVISE",
|
|
160
|
+
findings: [
|
|
161
|
+
{ class: "b", text: "Substrate-as-invariant: compiler must not name plugin_projector", severity: "P0" },
|
|
162
|
+
{ class: "a", text: "mode_name binding integrity across recompile not guaranteed", severity: "P0" },
|
|
163
|
+
{ class: "c", text: "Section ordering could improve readability", severity: "P2" }
|
|
164
|
+
]
|
|
165
|
+
},
|
|
166
|
+
{
|
|
167
|
+
id: "codex_gpt5.4",
|
|
168
|
+
label: "Codex (GPT-5.4)",
|
|
169
|
+
pool: "advisory",
|
|
170
|
+
verdict: "REJECT",
|
|
171
|
+
findings: [
|
|
172
|
+
{ class: "c", text: "Exhaustiveness of Inv-C3 not formally verifiable", severity: "P0" },
|
|
173
|
+
{ class: "c", text: "Missing rollback procedure specification", severity: "P0" },
|
|
174
|
+
{ class: "a", text: "Race condition possible if two compiles run concurrently", severity: "P1" }
|
|
175
|
+
]
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
id: "cursor_composer2.5",
|
|
179
|
+
label: "Cursor (Composer 2.5)",
|
|
180
|
+
pool: "advisory",
|
|
181
|
+
verdict: "REJECT",
|
|
182
|
+
findings: [
|
|
183
|
+
{ class: "c", text: "Design lacks complete error taxonomy", severity: "P0" },
|
|
184
|
+
{ class: "c", text: "No performance benchmarks specified", severity: "P1" },
|
|
185
|
+
{ class: "a", text: "File path validation insufficient for symlink traversal", severity: "P1" }
|
|
186
|
+
]
|
|
187
|
+
}
|
|
188
|
+
]
|
|
189
|
+
}
|
|
190
|
+
]
|
|
191
|
+
};
|
|
192
|
+
|
|
193
|
+
let currentData = null;
|
|
194
|
+
let currentRound = 0;
|
|
195
|
+
|
|
196
|
+
function loadSample() {
|
|
197
|
+
document.getElementById('json-input').value = JSON.stringify(SAMPLE, null, 2);
|
|
198
|
+
loadData();
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
function loadData() {
|
|
202
|
+
const errEl = document.getElementById('json-error');
|
|
203
|
+
errEl.textContent = '';
|
|
204
|
+
try {
|
|
205
|
+
currentData = JSON.parse(document.getElementById('json-input').value);
|
|
206
|
+
currentRound = 0;
|
|
207
|
+
renderRoundTabs();
|
|
208
|
+
renderRound(0);
|
|
209
|
+
document.getElementById('dashboard').style.display = 'block';
|
|
210
|
+
} catch (e) {
|
|
211
|
+
errEl.textContent = 'JSON parse error: ' + e.message;
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function renderRoundTabs() {
|
|
216
|
+
const tabs = document.getElementById('round-tabs');
|
|
217
|
+
tabs.innerHTML = '';
|
|
218
|
+
currentData.rounds.forEach((r, i) => {
|
|
219
|
+
const btn = document.createElement('button');
|
|
220
|
+
btn.className = 'round-tab' + (i === currentRound ? ' active' : '');
|
|
221
|
+
btn.textContent = 'Round ' + r.round;
|
|
222
|
+
btn.onclick = () => { currentRound = i; renderRoundTabs(); renderRound(i); };
|
|
223
|
+
tabs.appendChild(btn);
|
|
224
|
+
});
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
function renderRound(idx) {
|
|
228
|
+
const round = currentData.rounds[idx];
|
|
229
|
+
const reviewers = round.reviewers;
|
|
230
|
+
|
|
231
|
+
// Verdicts
|
|
232
|
+
const vEl = document.getElementById('consensus-verdicts');
|
|
233
|
+
vEl.innerHTML = reviewers.map(r =>
|
|
234
|
+
`<span style="margin-right:12px;">${r.label}: <span class="verdict ${r.verdict.toLowerCase()}">${r.verdict}</span></span>`
|
|
235
|
+
).join('<br>');
|
|
236
|
+
|
|
237
|
+
// Counts. A finding carrying carryover: true was raised in an earlier round
|
|
238
|
+
// and is still open; it does not enter the closing condition. An absent flag
|
|
239
|
+
// means new, which is what the round-1 shape and every earlier input produce.
|
|
240
|
+
let blocking = 0, carryover = 0, advisory = 0;
|
|
241
|
+
reviewers.forEach(r => r.findings.forEach(f => {
|
|
242
|
+
if (f.class === 'a' || f.class === 'b') {
|
|
243
|
+
if (f.carryover) carryover++; else blocking++;
|
|
244
|
+
} else advisory++;
|
|
245
|
+
}));
|
|
246
|
+
document.getElementById('blocking-count').textContent = blocking;
|
|
247
|
+
document.getElementById('carryover-count').textContent = carryover;
|
|
248
|
+
document.getElementById('advisory-count').textContent = advisory;
|
|
249
|
+
|
|
250
|
+
// Gate. The closing condition is new (a)+(b) = 0 — CLAUDE.md § Aggregation
|
|
251
|
+
// rule, L1 multi_llm_review_workflow § Convergence Rules. The vote tally is a
|
|
252
|
+
// reference value, shown and not required: the roster ratio can be
|
|
253
|
+
// arithmetically unreachable, and the intended close is (a)+(b) exhaustion
|
|
254
|
+
// declared by the operator. This panel therefore reports a freeze candidate,
|
|
255
|
+
// never a passed gate.
|
|
256
|
+
const blockingReviewers = reviewers.filter(r => r.pool === 'blocking');
|
|
257
|
+
const approveCount = blockingReviewers.filter(r => r.verdict === 'APPROVE').length;
|
|
258
|
+
const gateEl = document.getElementById('gate-status');
|
|
259
|
+
const votes = `votes ${approveCount}/${blockingReviewers.length} APPROVE (reference)`;
|
|
260
|
+
gateEl.innerHTML = blocking === 0
|
|
261
|
+
? `<span style="color:var(--approve);">✓ FREEZE CANDIDATE</span> — 0 new blocking P0, ${carryover} carryover still open — ${votes}`
|
|
262
|
+
: `<span style="color:var(--reject);">✗ NOT CLOSED</span> — ${blocking} new blocking P0 (a+b), ${carryover} carryover — ${votes}`;
|
|
263
|
+
|
|
264
|
+
// Bars
|
|
265
|
+
const barsEl = document.getElementById('finding-bars');
|
|
266
|
+
barsEl.innerHTML = '';
|
|
267
|
+
reviewers.forEach(r => {
|
|
268
|
+
const a = r.findings.filter(f => f.class === 'a').length;
|
|
269
|
+
const b = r.findings.filter(f => f.class === 'b').length;
|
|
270
|
+
const c = r.findings.filter(f => f.class === 'c').length;
|
|
271
|
+
const total = Math.max(a + b + c, 1);
|
|
272
|
+
const maxFindings = Math.max(...reviewers.map(rv => rv.findings.length), 1);
|
|
273
|
+
const scale = 100 / maxFindings;
|
|
274
|
+
barsEl.innerHTML += `<div class="bar-row">
|
|
275
|
+
<div class="bar-label">${r.id}</div>
|
|
276
|
+
<div class="bar-track">
|
|
277
|
+
<div class="bar-seg a" style="width:${a * scale}%" title="(a) ${a}"></div>
|
|
278
|
+
<div class="bar-seg b" style="width:${b * scale}%" title="(b) ${b}"></div>
|
|
279
|
+
<div class="bar-seg c" style="width:${c * scale}%" title="(c) ${c}"></div>
|
|
280
|
+
</div>
|
|
281
|
+
<span style="font-size:0.8rem;color:var(--text-muted);width:40px;">${a+b+c}</span>
|
|
282
|
+
</div>`;
|
|
283
|
+
});
|
|
284
|
+
|
|
285
|
+
// Table
|
|
286
|
+
const tbody = document.getElementById('findings-table');
|
|
287
|
+
tbody.innerHTML = '';
|
|
288
|
+
reviewers.forEach(r => {
|
|
289
|
+
r.findings.forEach(f => {
|
|
290
|
+
tbody.innerHTML += `<tr>
|
|
291
|
+
<td>${r.id}</td>
|
|
292
|
+
<td><span class="finding-tag finding-${f.class}">(${f.class})</span></td>
|
|
293
|
+
<td>${f.text}</td>
|
|
294
|
+
<td>${f.severity}</td>
|
|
295
|
+
</tr>`;
|
|
296
|
+
});
|
|
297
|
+
});
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
function copyAsPrompt() {
|
|
301
|
+
if (!currentData) return;
|
|
302
|
+
const round = currentData.rounds[currentRound];
|
|
303
|
+
let prompt = `## Multi-LLM Review Round ${round.round} Summary\n\n`;
|
|
304
|
+
prompt += `**Artifact:** ${currentData.artifact}\n\n`;
|
|
305
|
+
|
|
306
|
+
const blockingReviewers = round.reviewers.filter(r => r.pool === 'blocking');
|
|
307
|
+
const approves = blockingReviewers.filter(r => r.verdict === 'APPROVE').length;
|
|
308
|
+
prompt += `**Gate:** ${approves}/${blockingReviewers.length} blocking APPROVE\n\n`;
|
|
309
|
+
|
|
310
|
+
prompt += `### Blocking findings (a+b) — must address:\n\n`;
|
|
311
|
+
round.reviewers.forEach(r => {
|
|
312
|
+
r.findings.filter(f => f.class !== 'c').forEach(f => {
|
|
313
|
+
prompt += `- [${r.id}] (${f.class}) ${f.severity}: ${f.text}\n`;
|
|
314
|
+
});
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
prompt += `\n### Advisory findings (c) — reviewer preference:\n\n`;
|
|
318
|
+
round.reviewers.forEach(r => {
|
|
319
|
+
r.findings.filter(f => f.class === 'c').forEach(f => {
|
|
320
|
+
prompt += `- [${r.id}] (${f.class}) ${f.severity}: ${f.text}\n`;
|
|
321
|
+
});
|
|
322
|
+
});
|
|
323
|
+
|
|
324
|
+
navigator.clipboard.writeText(prompt).then(() => {
|
|
325
|
+
const btn = event.target;
|
|
326
|
+
btn.textContent = 'Copied!';
|
|
327
|
+
setTimeout(() => btn.textContent = 'Copy as Prompt', 1500);
|
|
328
|
+
});
|
|
329
|
+
}
|
|
330
|
+
</script>
|
|
331
|
+
</body>
|
|
332
|
+
</html>
|