kairos-chain 3.85.0 → 3.87.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +108 -0
- data/lib/kairos_mcp/version.rb +1 -1
- data/templates/knowledge/multi_llm_review_workflow/multi_llm_review_workflow.md +156 -31
- data/templates/skillsets/llm_client/lib/llm_client/bedrock_adapter.rb +17 -1
- data/templates/skillsets/llm_client/skillset.json +1 -1
- data/templates/skillsets/llm_client/test/test_bedrock_adapter_timeout.rb +57 -0
- data/templates/skillsets/model_provenance/hooks/observer.rb +413 -0
- data/templates/skillsets/model_provenance/hooks/transcript_reader.rb +238 -0
- data/templates/skillsets/model_provenance/lib/model_provenance/observation_store.rb +135 -0
- data/templates/skillsets/model_provenance/lib/model_provenance.rb +6 -0
- data/templates/skillsets/model_provenance/plugin/hooks.json +58 -0
- data/templates/skillsets/model_provenance/skillset.json +18 -0
- data/templates/skillsets/model_provenance/test/fixture_builder.rb +100 -0
- data/templates/skillsets/model_provenance/test/test_hook_command.rb +41 -0
- data/templates/skillsets/model_provenance/test/test_observation_store.rb +134 -0
- data/templates/skillsets/model_provenance/test/test_observer.rb +366 -0
- data/templates/skillsets/model_provenance/test/test_transcript_reader.rb +193 -0
- data/templates/skillsets/multi_llm_review/config/multi_llm_review.yml +26 -12
- data/templates/skillsets/multi_llm_review/lib/multi_llm_review/persona_assembly.rb +23 -2
- data/templates/skillsets/multi_llm_review/lib/multi_llm_review/provenance_binding.rb +110 -0
- data/templates/skillsets/multi_llm_review/lib/multi_llm_review/review_serializer.rb +2 -0
- data/templates/skillsets/multi_llm_review/skillset.json +1 -1
- data/templates/skillsets/multi_llm_review/test/test_provenance_binding.rb +227 -0
- data/templates/skillsets/multi_llm_review/tools/multi_llm_review.rb +3 -3
- data/templates/skillsets/multi_llm_review/tools/multi_llm_review_collect.rb +23 -2
- metadata +15 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: eac563609de17554382bdd2a4c2a42b21040fd60f43c4702eea153eeb5d5806b
|
|
4
|
+
data.tar.gz: 0cbb9994501dc8bdf9333e13f15bd25bc5c45e68f83abf1066e860606bc1c18a
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 4f41031abdbec48d77ed711a114a8be271321b4afee36e4a7b081550437afbf82b8a0a2aa91eba23fa3422873f8e5032202a00dc51a4cbb5758b72370a60d1dc
|
|
7
|
+
data.tar.gz: f0d09276a3259622de795ec1f7522e2b78af9c80c6934aa809195755f3fa8938d53996fbd1ab226e0884a5348383b92baf320a3c6c94e9d82ad94689d35cba87
|
data/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,114 @@ All notable changes to the `kairos-chain` gem will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [3.87.0] - 2026-09-24
|
|
8
|
+
|
|
9
|
+
### Added — `model_provenance` SkillSet: which model actually answered
|
|
10
|
+
|
|
11
|
+
Claude Code re-runs a request on another model when a safety classifier flags it
|
|
12
|
+
(from Opus 5.5 and the Fable models: biology → Opus 5, cyber → Opus 4.8) and the
|
|
13
|
+
session then stays on that model. The operator is told once in the main session and
|
|
14
|
+
not at all for subagents, and the model that took over reads the same system prompt
|
|
15
|
+
naming the model it replaced, so its account of itself cannot be trusted. Observed
|
|
16
|
+
on 2026-09-24: an audit subagent answered 90 of 102 records on `claude-opus-5`, with
|
|
17
|
+
nothing in its output saying so.
|
|
18
|
+
|
|
19
|
+
`model_provenance` is hooks only (no MCP tools, no core change). It derives the
|
|
20
|
+
answering model from what the harness recorded — per-response `message.model` and
|
|
21
|
+
the transcript's `fallback` marker — never from what a model says:
|
|
22
|
+
|
|
23
|
+
- **PostModelSwitch** records every model change; on `source: "auto"` it tells the
|
|
24
|
+
model which model it now is.
|
|
25
|
+
- **SubagentStop** records the subagent's latest run. A run starts only at a message
|
|
26
|
+
from the coordinator or a peer; Skill bodies (`turnCompanion` records) and the
|
|
27
|
+
agent's own task notifications are not resumes.
|
|
28
|
+
- **Stop** reads every finished response not yet reported — main transcript and every
|
|
29
|
+
subagent transcript, recursively — and shows one `[model_provenance] …` line per
|
|
30
|
+
affected transcript, including transcripts it could not read and subagents still
|
|
31
|
+
running. All observer state commits in one rename after the report is composed.
|
|
32
|
+
|
|
33
|
+
The transcript reader lives in `hooks/`; `lib/` exposes only the append-only
|
|
34
|
+
observation store. The projected command guards itself when neither
|
|
35
|
+
`KAIROS_DATA_DIR` nor `CLAUDE_PROJECT_DIR` is set (silent under Codex, a visible
|
|
36
|
+
non-blocking error elsewhere). Install: `kairos-chain skillset install
|
|
37
|
+
<gem>/templates/skillsets/model_provenance`, then re-project. Design:
|
|
38
|
+
`docs/model_provenance/`.
|
|
39
|
+
|
|
40
|
+
### Changed — `multi_llm_review` 0.11.0: persona seats can record the observed model
|
|
41
|
+
|
|
42
|
+
`orchestrator_reviews[]` entries take an optional `agent_id` (the id the Agent tool
|
|
43
|
+
returned for that persona's subagent). When any persona carries one, collect
|
|
44
|
+
resolves it through `model_provenance`'s store and the seat records
|
|
45
|
+
`model_source`, `model_observed`, `model_divergence` and `binding:
|
|
46
|
+
caller_declared`, with each persona's observation in its `persona_rows` entry. A
|
|
47
|
+
persona is divergent when any observed response came from a model other than the
|
|
48
|
+
declared persona model; the seat is observed only when every persona's observation
|
|
49
|
+
is complete. Without any `agent_id` the record is exactly as before.
|
|
50
|
+
|
|
51
|
+
### Changed — review roster: Opus 5 → Opus 5.5, Fable 5 → Fable 5.1
|
|
52
|
+
|
|
53
|
+
The rotating frontier slot is `claude-opus-5-5` (label `claude_cli_opus5.5`) and
|
|
54
|
+
the reserve slot `claude-fable-5-1`. Opus 5.5 is the one model that runs at medium
|
|
55
|
+
when effort is omitted, so the explicit `high` in `effort_map` is what keeps it at
|
|
56
|
+
high. L1 `multi_llm_review_workflow` 3.14.0 → 3.15.0: prescriptive text only;
|
|
57
|
+
measurements and history that name Opus 5 or Fable 5 stay as written.
|
|
58
|
+
|
|
59
|
+
## [3.86.0] - 2026-09-21
|
|
60
|
+
|
|
61
|
+
### Changed — multi-LLM review: a per-round check for which review you are in
|
|
62
|
+
|
|
63
|
+
Templates only: L1 `multi_llm_review_workflow` 3.13.0 → 3.14.0. No library code,
|
|
64
|
+
no config.
|
|
65
|
+
|
|
66
|
+
**The rule.** § Step -1 gains rule 7: classify each round's findings by CATEGORY,
|
|
67
|
+
not only by the (a)/(b)/(c) severity that rule 1 and § Convergence Rules already
|
|
68
|
+
require. Structural gaps ("this cannot work") and fix correctness ("the fix is
|
|
69
|
+
wrong") are design-review categories; missing wiring ("this does not work") is an
|
|
70
|
+
implementation-review category. When a design loop's findings have moved to the
|
|
71
|
+
implementation category, the design review is over — freeze, hand the rest to the
|
|
72
|
+
implementation queue, and do not dispatch another round.
|
|
73
|
+
|
|
74
|
+
**Why it is new.** The content is not new; the *step* is. `multi_llm_reviewer_
|
|
75
|
+
evaluation` § Bug Category Differentiation Across Rounds and this document's
|
|
76
|
+
§ Convergence Curve have both described the progression for months, as
|
|
77
|
+
observations. Nothing told an orchestrator to perform the check each round, so a
|
|
78
|
+
loop could run eight rounds without anyone asking which review it was in.
|
|
79
|
+
|
|
80
|
+
**The loop that paid for it.** GenomicsChain service (2.5) provenance anchoring,
|
|
81
|
+
2026-09-20/21, 8 rounds, orchestrator Opus 5. Rounds 1–4 exhausted the
|
|
82
|
+
design-category findings — the record contract, the verifier's readability, the
|
|
83
|
+
proof TTL, the packaging. Rounds 5–8 then spent four rounds on a single
|
|
84
|
+
implementation-category class (a malformed bundle outgrading an honest one, in
|
|
85
|
+
the adapter's reading of broken input) and produced +102 lines of design against
|
|
86
|
+
+271 of code and +397 of tests, while the reviewed artifact grew 56 KB → 180 KB.
|
|
87
|
+
|
|
88
|
+
Two things about that are worth keeping. First, the orchestrator's (a)/(b)/(c)
|
|
89
|
+
trend table looked healthy the whole time — blocking rows 15 → 12 → 8 → 7 → 7 →
|
|
90
|
+
9 → 10 → 6, density 26.7 → 3.3 per 100 KB — because the findings *were*
|
|
91
|
+
narrowing. Severity and category are different axes, and only one of them was
|
|
92
|
+
being read. Second, the surface grew because rule 2 was being broken: the code
|
|
93
|
+
sat under an `APPENDIX (advisory only — findings against these never block)`
|
|
94
|
+
header, the seats honoured it (round 8's codex labelled two of its three findings
|
|
95
|
+
"Advisory under the appendix rule") and the orchestrator treated them as blocking
|
|
96
|
+
anyway. Rule 2 exists to collapse the claim surface; a surface that grows every
|
|
97
|
+
round is the tell that it is not being kept.
|
|
98
|
+
|
|
99
|
+
The loop ended on the operator's observation, not on any recorded signal: *the
|
|
100
|
+
adapter is an implementation problem, and a design review that runs tests has
|
|
101
|
+
stopped being a design review.*
|
|
102
|
+
|
|
103
|
+
**Also recorded in § Experimental Data**, because the same loop dropped them: the
|
|
104
|
+
≤5 fixes-per-round cap (rule 3) ran at 7/6/6/6, the pre-flight falsifier (rule 4)
|
|
105
|
+
was skipped for three consecutive rounds, the per-round
|
|
106
|
+
`reviewer_evaluation_observation_<reviewer>_<date>` records (§ L2 Save Points)
|
|
107
|
+
were never written, and no seat was asked for a closure verdict on its own
|
|
108
|
+
prior-round P0s (§ Convergence Rules). A loop that drops five of this document's
|
|
109
|
+
rules at once is not a loop that ran out of rules to follow.
|
|
110
|
+
|
|
111
|
+
Records (GenomicsChain_SkillSets): L2 `decision_design_frozen_prov_anchoring_20260921`,
|
|
112
|
+
`reviewer_evaluation_observation_prov_anchoring_loop_20260921`, and the round
|
|
113
|
+
records `review_r5_…` through `review_r8_prov_anchoring_v0_1_9_20260921`.
|
|
114
|
+
|
|
7
115
|
## [3.85.0] - 2026-09-06
|
|
8
116
|
|
|
9
117
|
### Changed — multi-LLM review: high effort by default, gpt-6-astra seats, gpt-5.5 retires
|
data/lib/kairos_mcp/version.rb
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: multi_llm_review_workflow
|
|
3
3
|
description: "Multi-LLM review methodology and execution — workflow pattern, CLI tooling, consensus analysis, Persona Assembly. Applicable to design, implementation, documentation, or any artifact."
|
|
4
|
-
version: "3.
|
|
4
|
+
version: "3.15.0"
|
|
5
5
|
tags:
|
|
6
6
|
- workflow
|
|
7
7
|
- review
|
|
@@ -69,6 +69,42 @@ write a review spec and declare it frozen for the round:
|
|
|
69
69
|
6. **Reference originals by path + sha256; do not transcribe.** Reviewers
|
|
70
70
|
read the repository; the artifact carries the manifest. Transcription
|
|
71
71
|
errors are undetectable and 100KB+ pastes rot.
|
|
72
|
+
7. **Check which review you are in, every round — category, not severity.**
|
|
73
|
+
Classify each round's findings by CATEGORY against
|
|
74
|
+
`multi_llm_reviewer_evaluation` § Bug Category Differentiation Across
|
|
75
|
+
Rounds, in addition to the (a)/(b)/(c) severity classification rule 1 and
|
|
76
|
+
§ Convergence Rules already require. Structural gaps ("this cannot work")
|
|
77
|
+
and fix correctness ("the fix is wrong") are design-review categories;
|
|
78
|
+
missing wiring ("this does not work") is an implementation-review
|
|
79
|
+
category. **When a design loop's findings have moved to the implementation
|
|
80
|
+
category, the design review is over**: freeze the design, hand the
|
|
81
|
+
remaining findings to the implementation queue, and do not dispatch
|
|
82
|
+
another round. Severity and category are different axes — a trend of
|
|
83
|
+
narrowing (a)/(b) findings says the loop is healthy, and says nothing
|
|
84
|
+
about whether it is still reviewing the design.
|
|
85
|
+
|
|
86
|
+
Two cheap tells that a design loop has crossed over, both readable
|
|
87
|
+
without any new instrumentation: the round's output is mostly code and
|
|
88
|
+
tests rather than design text, and the claim surface is growing instead
|
|
89
|
+
of collapsing (rule 2 exists to collapse it; if it is growing, rule 2 is
|
|
90
|
+
being broken somewhere).
|
|
91
|
+
|
|
92
|
+
Evidence — GenomicsChain service (2.5) provenance anchoring, 2026-09-20/21,
|
|
93
|
+
8 rounds, orchestrator Opus 5. Rounds 5–8 found exactly one class, a
|
|
94
|
+
malformed bundle outgrading an honest one, entirely in the adapter's
|
|
95
|
+
reading of broken input: the implementation category. Those four rounds
|
|
96
|
+
produced +102 lines of design against +271 of code and +397 of tests, and
|
|
97
|
+
the artifact grew 56 KB → 180 KB. Rule 2 was broken throughout — the code
|
|
98
|
+
sat under an APPENDIX header reading "advisory only, never block", the
|
|
99
|
+
seats obeyed it (round 8's codex marked two of its three findings
|
|
100
|
+
"Advisory under the appendix rule") and the orchestrator treated those
|
|
101
|
+
findings as blocking anyway. The orchestrator ran the (a)/(b)/(c) trend
|
|
102
|
+
table every round and it looked healthy, because the findings WERE
|
|
103
|
+
narrowing; what it never asked was which review the categories described.
|
|
104
|
+
The operator stopped the loop by naming the category shift directly — the
|
|
105
|
+
adapter is an implementation problem, and a design review that runs tests
|
|
106
|
+
has stopped being a design review. Records: L2
|
|
107
|
+
`decision_design_frozen_prov_anchoring_20260921`.
|
|
72
108
|
|
|
73
109
|
Corollaries observed in the same loop: fix the *class*, and fix every copy —
|
|
74
110
|
a corrected lib comment whose refuted twin survives in a test file costs a
|
|
@@ -461,9 +497,9 @@ they disagree, the config is right and this section is stale.
|
|
|
461
497
|
- [ ] Agent Team Personas model: = orchestrator model (NOT a different model),
|
|
462
498
|
unless you are running personas on a different model on purpose — then
|
|
463
499
|
it is whatever you declare as persona_model (see § Persona execution model)
|
|
464
|
-
- [ ] Subprocess CLI model: Opus 4.6. The other Claude roster slot is Opus 5,
|
|
500
|
+
- [ ] Subprocess CLI model: Opus 4.6. The other Claude roster slot is Opus 5.5,
|
|
465
501
|
which under the default "delegate" strategy is taken by your persona team
|
|
466
|
-
rather than spawned — so when you are Opus 5, Opus 4.6 is the only Claude
|
|
502
|
+
rather than spawned — so when you are Opus 5.5, Opus 4.6 is the only Claude
|
|
467
503
|
CLI subprocess
|
|
468
504
|
- [ ] Codex model: gpt-6-astra, with -m. One codex slot since gpt-5.5 was
|
|
469
505
|
retired 2026-09-05 — do not add a second codex entry expecting the old
|
|
@@ -818,16 +854,17 @@ outside this repository — see the incident recorded in § Pre-flight checklist
|
|
|
818
854
|
| **Cursor Agent** | `agent -p --model composer-2.5` | File reference (stdin NOT supported) | stdout redirect: `> output.md` | composer-2.5, passed explicitly — never relying on the CLI default |
|
|
819
855
|
| **Claude Code** | Agent tool (internal) | Direct prompt string | Write to workspace file | Orchestrator model, or the declared `persona_model` when personas run elsewhere |
|
|
820
856
|
| **Claude CLI (4.6)** | `claude -p --model claude-opus-4-6` | stdin pipe: `cat prompt.md \| claude -p --model claude-opus-4-6` | stdout redirect: `> output.md` | Opus 4.6 — the calibrated anchor, deliberately not a frontier model |
|
|
821
|
-
| **Claude CLI (frontier)** | `claude -p --model claude-opus-5` | stdin pipe: `cat prompt.md \| claude -p --model claude-opus-5` | stdout redirect: `> output.md` | Runs only when Opus 5 is *not* the orchestrator; when it is, that slot is taken by the persona team |
|
|
857
|
+
| **Claude CLI (frontier)** | `claude -p --model claude-opus-5-5` | stdin pipe: `cat prompt.md \| claude -p --model claude-opus-5-5` | stdout redirect: `> output.md` | Runs only when Opus 5.5 is *not* the orchestrator; when it is, that slot is taken by the persona team |
|
|
822
858
|
|
|
823
859
|
`--bare` must NOT be passed (established 2026-07-23): it skips credential
|
|
824
860
|
loading and the subprocess fails with "Not logged in". The project-instruction
|
|
825
861
|
bias it was meant to suppress is handled by `review_context: independent`
|
|
826
862
|
instead.
|
|
827
863
|
|
|
828
|
-
Fable
|
|
829
|
-
after five consecutive non-substantive returns
|
|
830
|
-
|
|
864
|
+
Fable appears in no row above. Fable 5 was retired from the roster on 2026-07-26
|
|
865
|
+
after five consecutive non-substantive returns; the reserve container now holds
|
|
866
|
+
its successor, Fable 5.1 (since 2026-09-24) — see § Reserve observers
|
|
867
|
+
(escalation).
|
|
831
868
|
|
|
832
869
|
### Thinking Effort Configuration (validated 2026-04-20)
|
|
833
870
|
|
|
@@ -835,11 +872,11 @@ Based on cross-evaluation experiment (7 models × 4 tasks + Nomic, 518 CLI calls
|
|
|
835
872
|
|
|
836
873
|
| Role | Model | Effort Flag | Rationale |
|
|
837
874
|
|------|-------|-------------|-----------|
|
|
838
|
-
| **Primary (orchestrator)** | session default |
|
|
839
|
-
| **Reviewer: Agent Team** | = orchestrator, or the declared `persona_model` |
|
|
875
|
+
| **Primary (orchestrator)** | session default | Claude Code's own setting | Not set by this config — record the level actually in effect (see the 2026-09-24 note) |
|
|
876
|
+
| **Reviewer: Agent Team** | = orchestrator, or the declared `persona_model` | the session's level, unless the persona's agent file sets `effort:` | Personas inherit whichever model — and effort — actually runs them |
|
|
840
877
|
| **Reviewer: Claude CLI** | Opus 4.6, plus any frontier roster slot the orchestrator is not | `--effort high` (config `effort: high`) | Operator instruction 2026-09-05; supersedes the 2026-04-29 default-effort policy — see the note below the table |
|
|
841
|
-
| **Coding sub-agent** | Opus 5 | `--effort
|
|
842
|
-
| **Design sub-agent** | Opus 5 | `--effort high` |
|
|
878
|
+
| **Coding sub-agent** | Opus 5.5 | `--effort high` | Operator default for KairosChain (2026-09-24); not measured here (see note) |
|
|
879
|
+
| **Design sub-agent** | Opus 5.5 | `--effort high` | Operator default for KairosChain (2026-09-24); not measured here (see note) |
|
|
843
880
|
| **Codex** | GPT-6-astra / GPT-5.5 | `-c model_reasoning_effort=high` | Same operator instruction. The earlier "(no flag) / fixed effort" entry was wrong: codex_adapter has always emitted this flag when the roster set `effort` |
|
|
844
881
|
| **Cursor Agent** | Composer-2.5 | (no flag) | Genuinely has no effort control — cursor_adapter builds no such flag, so an `effort:` key on a cursor roster entry is recorded and never sent |
|
|
845
882
|
|
|
@@ -871,6 +908,20 @@ a starting point to sweep down from, not a validated setting. Separately, the
|
|
|
871
908
|
for its ambiguity-preserving bias, not for capability, so a frontier successor
|
|
872
909
|
does not replace it.
|
|
873
910
|
|
|
911
|
+
Note (2026-09-24): the frontier rows moved from Opus 5 to Opus 5.5 (operator
|
|
912
|
+
instruction — 5.5 is now Claude Code's default orchestrator and takes over
|
|
913
|
+
every seat Opus 5 held), and both sub-agent rows are set to `high`, the
|
|
914
|
+
operator's default effort for KairosChain. Two facts about 5.5 matter when
|
|
915
|
+
reading any effort in this document. It is the one model that runs at
|
|
916
|
+
**medium** when effort is omitted (every other model defaults to high), so
|
|
917
|
+
every seat that should run at high has to say so. And level names do not mean
|
|
918
|
+
the same amount of thinking across models — Anthropic reports 5.5 at medium
|
|
919
|
+
matching Opus 5 at high — so equal settings across seats are not equal
|
|
920
|
+
thinking. The orchestrator and a persona team run at the Claude Code session's
|
|
921
|
+
level (the Agent tool has no effort parameter; an agent file's `effort:` line
|
|
922
|
+
overrides it), so a session at xhigh puts the persona seat at xhigh while the
|
|
923
|
+
CLI seats run at high. Record both levels in the round.
|
|
924
|
+
|
|
874
925
|
Key findings:
|
|
875
926
|
- **Opus 4.6** high effort improves Evaluator/Strategy (+0.43/+0.200 Nomic), not Response
|
|
876
927
|
- **Opus 4.7** high effort improves Response/Thinking (+0.81 code, +0.53 philosophy), not Evaluator
|
|
@@ -908,11 +959,11 @@ record no longer claims a model that did not answer.
|
|
|
908
959
|
**Rule**: When invoking `multi_llm_review` (or running this workflow manually), the
|
|
909
960
|
orchestrating LLM MUST pass its own model identifier as `orchestrator_model`.
|
|
910
961
|
|
|
911
|
-
**Rationale**: The reviewer roster contains more than one Claude entry (Opus 5
|
|
912
|
-
and Opus 4.6 as of 2026-07-26). A frontier slot sits in the roster on purpose:
|
|
962
|
+
**Rationale**: The reviewer roster contains more than one Claude entry (Opus 5.5
|
|
963
|
+
and Opus 4.6 as of 2026-09-24; Opus 5 held the frontier entry from 2026-07-26). A frontier slot sits in the roster on purpose:
|
|
913
964
|
when it is the current session model it matches `orchestrator_model` and becomes
|
|
914
965
|
the persona-team slot; when it is not, it is dispatched as a `claude -p`
|
|
915
|
-
subprocess. With the current two-entry Claude side, an Opus 5 orchestrator leaves
|
|
966
|
+
subprocess. With the current two-entry Claude side, an Opus 5.5 orchestrator leaves
|
|
916
967
|
Opus 4.6 as the only Claude CLI subprocess. No per-orchestrator config branch is
|
|
917
968
|
needed either way.
|
|
918
969
|
To avoid the orchestrator reviewing its
|
|
@@ -924,7 +975,7 @@ composition adapts automatically.
|
|
|
924
975
|
|
|
925
976
|
**Why "argument-passing" not "file-introspection"**:
|
|
926
977
|
- The orchestrator's model identity lives in *its own context* (system prompt
|
|
927
|
-
declares e.g. "You are powered by Fable 5"). No external file or env var is
|
|
978
|
+
declares e.g. "You are powered by Fable 5.1"). No external file or env var is
|
|
928
979
|
authoritative — `/model` switches change context immediately.
|
|
929
980
|
- MCP protocol does not transmit caller-model info; only the orchestrator can
|
|
930
981
|
truthfully report its own identity. This is genuine self-reference: the system
|
|
@@ -936,8 +987,8 @@ composition adapts automatically.
|
|
|
936
987
|
**How orchestrator obtains its model ID**:
|
|
937
988
|
- Claude Code sessions: read the system prompt line "You are powered by the
|
|
938
989
|
model named ... The exact model ID is ...". Use the exact ID as stated,
|
|
939
|
-
whatever its form (e.g. `claude-opus-5`, `claude-fable-5`). Strip any context
|
|
940
|
-
suffix — `claude-opus-5[1m]` is passed as `claude-opus-5`, since the roster
|
|
990
|
+
whatever its form (e.g. `claude-opus-5-5`, `claude-fable-5-1`). Strip any context
|
|
991
|
+
suffix — `claude-opus-5-5[1m]` is passed as `claude-opus-5-5`, since the roster
|
|
941
992
|
matches on the bare model ID.
|
|
942
993
|
- Other hosts: use whatever introspection the host provides; if none, pass
|
|
943
994
|
`null` and accept that no exclusion happens.
|
|
@@ -947,7 +998,7 @@ composition adapts automatically.
|
|
|
947
998
|
multi_llm_review(
|
|
948
999
|
artifact_path: "log/design.md",
|
|
949
1000
|
review_type: "design",
|
|
950
|
-
orchestrator_model: "claude-opus-5"
|
|
1001
|
+
orchestrator_model: "claude-opus-5-5" # MUST be set by caller; bare ID, no [1m] suffix
|
|
951
1002
|
)
|
|
952
1003
|
```
|
|
953
1004
|
|
|
@@ -966,7 +1017,7 @@ multi_llm_review(
|
|
|
966
1017
|
model: the first is taken over by the persona team, the second leaves as the
|
|
967
1018
|
caller's own, and any beyond that run — as fresh external processes on that
|
|
968
1019
|
model, which is what the "subprocess" strategy buys deliberately. On the
|
|
969
|
-
current roster (one Opus 5 entry) the question does not arise. Settled
|
|
1020
|
+
current roster (one Opus 5.5 entry) the question does not arise. Settled
|
|
970
1021
|
2026-07-27: the invariant fixes a count for the persona clause and states none
|
|
971
1022
|
for this one, and the shipped behaviour follows the text rather than widening
|
|
972
1023
|
it. A caller who wants a same-model slot to run anyway asks for it with
|
|
@@ -1009,6 +1060,18 @@ cross-model subprocess reviewers give epistemic diversity. The two are complemen
|
|
|
1009
1060
|
performance/api-design; doc → ontologist/skeptic/integration)
|
|
1010
1061
|
- Collect persona results: each as `{persona, verdict (APPROVE|REVISE|REJECT),
|
|
1011
1062
|
reasoning, findings: [{severity, issue}, ...]}`
|
|
1063
|
+
- Bind each persona to its subagent: add `agent_id` — the id the Agent tool
|
|
1064
|
+
returned for that persona's subagent — to its entry. With the model_provenance
|
|
1065
|
+
SkillSet installed, the seat then records the model the harness observed
|
|
1066
|
+
answering, not only the declared one: a safety-classifier fallback inside a
|
|
1067
|
+
persona (Opus 5.5 → Opus 5 on biology, → Opus 4.8 on cyber) otherwise leaves
|
|
1068
|
+
one model's findings recorded under another's name. The binding is your
|
|
1069
|
+
declaration and is recorded as such; a persona without one, or whose id
|
|
1070
|
+
resolves to no record, is recorded as unobserved with its cause.
|
|
1071
|
+
- Call collect after the personas' task-notifications, not on their
|
|
1072
|
+
hand-backs: the hand-back reaches you 1.2–13.6 s before the subagent's final
|
|
1073
|
+
record (10 of 10 observed, 2026-09-24), and a persona collected before its
|
|
1074
|
+
record exists is recorded as unobserved.
|
|
1012
1075
|
|
|
1013
1076
|
**Call 2**: `multi_llm_review_collect(collect_token, orchestrator_reviews: [...])`
|
|
1014
1077
|
- Persona Assembly: any REJECT → REJECT; else any REVISE → REVISE; else APPROVE
|
|
@@ -1043,8 +1106,8 @@ the caller.
|
|
|
1043
1106
|
|
|
1044
1107
|
```
|
|
1045
1108
|
multi_llm_review(
|
|
1046
|
-
orchestrator_model: "claude-fable-5",
|
|
1047
|
-
persona_model: "claude-opus-5",
|
|
1109
|
+
orchestrator_model: "claude-fable-5-1", # who is calling
|
|
1110
|
+
persona_model: "claude-opus-5-5", # who the personas actually run on
|
|
1048
1111
|
...
|
|
1049
1112
|
)
|
|
1050
1113
|
```
|
|
@@ -1087,8 +1150,9 @@ Points worth knowing before using it:
|
|
|
1087
1150
|
- Reserve entries obey INV-E5 exactly like roster entries: they name their
|
|
1088
1151
|
provider and model, and inherit no CLI default.
|
|
1089
1152
|
|
|
1090
|
-
As of 2026-
|
|
1091
|
-
after five consecutive
|
|
1153
|
+
As of 2026-09-24 the container holds Fable 5.1. Its predecessor Fable 5 held it
|
|
1154
|
+
from 2026-07-26, after being retired from the roster for five consecutive
|
|
1155
|
+
non-substantive returns; Fable 5.1 has not yet run in this slot.
|
|
1092
1156
|
|
|
1093
1157
|
### Substance and the denominator
|
|
1094
1158
|
|
|
@@ -1285,7 +1349,7 @@ readable until GC. Read them directly and synthesize manually, then re-run
|
|
|
1285
1349
|
before, silently swapping the model behind an unchanged role label
|
|
1286
1350
|
- **Codex workspace**: `-C /path/to/workspace` to set working directory
|
|
1287
1351
|
- **Claude Agent paths**: Write within workspace (e.g., `log/`), not `/tmp`
|
|
1288
|
-
- **Claude CLI (Opus 4.6 / any non-orchestrator frontier slot)**: `claude -p --model claude-opus-4-6` (likewise `claude-opus-5`) runs as external process. Uses stdin pipe (like Codex). Do NOT pass `--bare` — it skips credential loading and the subprocess dies with "Not logged in" (established 2026-07-23). Project-instruction bias is suppressed via `review_context: independent`, not via `--bare`
|
|
1352
|
+
- **Claude CLI (Opus 4.6 / any non-orchestrator frontier slot)**: `claude -p --model claude-opus-4-6` (likewise `claude-opus-5-5`) runs as external process. Uses stdin pipe (like Codex). Do NOT pass `--bare` — it skips credential loading and the subprocess dies with "Not logged in" (established 2026-07-23). Project-instruction bias is suppressed via `review_context: independent`, not via `--bare`
|
|
1289
1353
|
- **Claude CLI parallelism**: Agent tool (internal, orchestrator model) + Bash `claude -p` (external, Opus 4.6 and any other Claude roster slot the orchestrator is not) run truly in parallel as separate processes
|
|
1290
1354
|
- **Claude CLI file access**: a plain `claude -p` review subprocess should not need file access. Ensure the review prompt includes all artifact content inline (rule #6). Use `--add-dir` + `--allowedTools "Read,Glob,Grep"` if file access is genuinely needed, and accept that CLAUDE.md is loaded (the old `--bare` workaround is unusable — see above)
|
|
1291
1355
|
|
|
@@ -1437,10 +1501,10 @@ Step 3: Execute the configured roster in parallel (currently 4 slots, one of
|
|
|
1437
1501
|
- Bash(background): cat prompt.md | codex exec -m gpt-6-astra -c model_reasoning_effort=high -C workspace -o log/review_codex_gpt6-astra.md -
|
|
1438
1502
|
- Bash(background): agent -p --trust --model composer-2.5 "Read prompt and review..." > log/review_cursor.md
|
|
1439
1503
|
(no effort flag — Cursor has no effort control)
|
|
1440
|
-
- Agent(background): Claude Team (orchestrator model, e.g. Opus 5) → write to log/review_claude_team_opus5.md
|
|
1504
|
+
- Agent(background): Claude Team (orchestrator model, e.g. Opus 5.5) → write to log/review_claude_team_opus5.5.md
|
|
1441
1505
|
- Bash(background): cat prompt.md | claude -p --model claude-opus-4-6 --effort high > log/review_claude_opus4.6.md 2>log/review_claude_opus4.6.stderr.log
|
|
1442
|
-
(add a line per further Claude roster slot you are not; with the 2026-
|
|
1443
|
-
roster an Opus 5 orchestrator has none, so opus-4.6 is the only one)
|
|
1506
|
+
(add a line per further Claude roster slot you are not; with the 2026-09-24
|
|
1507
|
+
roster an Opus 5.5 orchestrator has none, so opus-4.6 is the only one)
|
|
1444
1508
|
|
|
1445
1509
|
Step 4: Collect and validate
|
|
1446
1510
|
- Wait for all to complete (background task notifications)
|
|
@@ -1477,10 +1541,10 @@ log/{artifact}_review{N}_{llm_id}_{date}.md # Individual reviews
|
|
|
1477
1541
|
log/{artifact}_review{N}_consensus_{date}.md # Consensus analysis
|
|
1478
1542
|
```
|
|
1479
1543
|
|
|
1480
|
-
LLM identifiers: `claude_cli_opus5`, `claude_cli_opus4.6`,
|
|
1544
|
+
LLM identifiers: `claude_cli_opus5.5`, `claude_cli_opus4.6`,
|
|
1481
1545
|
`codex_gpt6-astra`, `cursor_composer2.5`, `cursor_gpt5.4`,
|
|
1482
1546
|
`cursor_premium`. The delegated slot is reported as `claude_team_<model>`
|
|
1483
|
-
(e.g. `claude_team_claude-opus-5`), assembled at collect time — the roster's
|
|
1547
|
+
(e.g. `claude_team_claude-opus-5-5`), assembled at collect time — the roster's
|
|
1484
1548
|
own labels stay CLI-neutral because either frontier entry can take either path.
|
|
1485
1549
|
(legacy, pre-2026-06-10: `claude_opus4.6`, `claude_team_opus4.6`, `claude_team_opus4.7`,
|
|
1486
1550
|
`claude_cli_opus4.7`, `cursor_composer2`; retired 2026-07-23: `codex_gpt5.4`;
|
|
@@ -1488,7 +1552,8 @@ retired 2026-07-25: `claude_cli_opus4.8`, `claude_team_fable5`;
|
|
|
1488
1552
|
retired 2026-07-26: `claude_cli_fable5` — five consecutive non-substantive
|
|
1489
1553
|
returns, 85-128 characters in 5-7 seconds, no findings and no verdict text;
|
|
1490
1554
|
retired 2026-09-05: `codex_gpt5.6-sol`, replaced by `codex_gpt6-astra`, and
|
|
1491
|
-
`codex_gpt5.5`, not replaced
|
|
1555
|
+
`codex_gpt5.5`, not replaced; retired 2026-09-24: `claude_cli_opus5`, succeeded in
|
|
1556
|
+
the same slot by `claude_cli_opus5.5`. Runs recorded under a retired identifier keep it —
|
|
1492
1557
|
the label names the model that answered, so renaming old records would attribute
|
|
1493
1558
|
one model's findings to another)
|
|
1494
1559
|
|
|
@@ -1825,5 +1890,65 @@ Compression ratio: parallel agent raw → Assembly ≈ 2:1
|
|
|
1825
1890
|
by (a)+(b) exhaustion without ever reaching it. Every ratio in this document
|
|
1826
1891
|
is a reference figure.
|
|
1827
1892
|
|
|
1893
|
+
- Step -1 rule 7 added, the per-round category check (v3.14.0, 2026-09-21,
|
|
1894
|
+
operator instruction). The rule the loop below needed already existed as an
|
|
1895
|
+
observation — `multi_llm_reviewer_evaluation` § Bug Category Differentiation
|
|
1896
|
+
and this document's § Convergence Curve both describe the progression — but
|
|
1897
|
+
neither was a step anyone was told to perform each round, so eight rounds ran
|
|
1898
|
+
without it. GenomicsChain service (2.5) provenance anchoring: R1–R4 exhausted
|
|
1899
|
+
the design-category findings (record contract, verifier readability, proof
|
|
1900
|
+
TTL, packaging); R5–R8 then spent four rounds on one implementation-category
|
|
1901
|
+
class, produced +102 design lines against +271 code and +397 test lines, and
|
|
1902
|
+
grew the artifact 56 KB → 180 KB while rule 2 was being broken. The
|
|
1903
|
+
orchestrator's own trend table showed healthy narrowing the whole time, which
|
|
1904
|
+
is the point: severity narrowed while the category had already changed.
|
|
1905
|
+
The loop closed on the operator's observation, not on any recorded signal.
|
|
1906
|
+
Also recorded, because the same loop dropped them: the ≤5 fixes-per-round cap
|
|
1907
|
+
(rule 3) went to 7/6/6/6, the pre-flight falsifier (rule 4) was skipped for
|
|
1908
|
+
three consecutive rounds, per-round
|
|
1909
|
+
`reviewer_evaluation_observation_<reviewer>_<date>` records (§ L2 Save Points)
|
|
1910
|
+
were never written, and no seat was asked for a closure verdict on its own
|
|
1911
|
+
prior-round P0s (§ Convergence Rules). A loop that drops five of this
|
|
1912
|
+
document's rules at once is not a loop that ran out of rules to follow.
|
|
1913
|
+
Records: L2 `decision_design_frozen_prov_anchoring_20260921`,
|
|
1914
|
+
`review_r5_prov_anchoring_v0_1_5_20260921` .. `review_r8_prov_anchoring_v0_1_9_20260921`,
|
|
1915
|
+
report `docs/reports/prov_anchoring_mlr_status_20260921/report.html` (GenomicsChain_SkillSets)
|
|
1916
|
+
|
|
1917
|
+
- Persona seats can record the observed model (v3.15.0, 2026-09-24, operator
|
|
1918
|
+
instruction). `orchestrator_reviews[]` entries take an optional `agent_id`;
|
|
1919
|
+
when any is supplied, collect resolves each through the model_provenance
|
|
1920
|
+
SkillSet's observation store and the seat carries `model_source`,
|
|
1921
|
+
`model_observed`, `model_divergence` and `binding: caller_declared`, with
|
|
1922
|
+
each persona's observation in its `persona_rows` entry. A persona is
|
|
1923
|
+
divergent when any observed response came from a model other than the
|
|
1924
|
+
declared persona model; the seat takes the most severe state (divergent >
|
|
1925
|
+
unobserved > matching) and is observed only when every persona's observation
|
|
1926
|
+
is complete. Without any `agent_id` the record is exactly as before. Why:
|
|
1927
|
+
Claude Code re-runs a flagged request on a fallback model and stays there,
|
|
1928
|
+
silently for subagents — observed twice on 2026-09-24 (62 and 208 responses,
|
|
1929
|
+
56 and 148 of them on Opus 5 after the switch). Design:
|
|
1930
|
+
`docs/model_provenance/design_v0.3.md`.
|
|
1931
|
+
|
|
1932
|
+
- Opus 5 → Opus 5.5 and Fable 5 → Fable 5.1 (v3.14.1, 2026-09-24, operator
|
|
1933
|
+
instruction). Opus 5.5 is now Claude Code's default orchestrator, so it takes
|
|
1934
|
+
over every seat Opus 5 held and nothing else about the roster changes: still
|
|
1935
|
+
4 seats (Opus 5.5 as orchestrator and persona team, Opus 4.6 CLI, codex
|
|
1936
|
+
gpt-6-astra, cursor composer-2.5), still `high` on every seat with an effort
|
|
1937
|
+
control. The frontier slot's label becomes `claude_cli_opus5.5`; the reserve
|
|
1938
|
+
container's occupant becomes Fable 5.1 (`claude_cli_fable5.1`). Only text
|
|
1939
|
+
that prescribes current behaviour was changed. Measurements, seat profiles and
|
|
1940
|
+
history that name Opus 5 or Fable 5 stay as written, because they are facts
|
|
1941
|
+
about those models — in particular the `claude_team_opus-5` persona row in
|
|
1942
|
+
§ Step 0.1 is Opus 5's record, not a profile for 5.5. Why the
|
|
1943
|
+
config's orchestrator slot had to move and not merely be documented: the slot
|
|
1944
|
+
is matched to `orchestrator_model` by exact string, so a 5.5 orchestrator
|
|
1945
|
+
against a `claude-opus-5` slot took over nothing — its persona team joined as
|
|
1946
|
+
a fifth observer and the Opus 5 slot still ran as a CLI subprocess (found in
|
|
1947
|
+
the 2026-09-24 Opus 5.5 guidance audit). Effort: see the 2026-09-24 note under
|
|
1948
|
+
§ Thinking Effort Configuration, including that the persona seat follows the
|
|
1949
|
+
session's level while CLI seats follow the config.
|
|
1950
|
+
|
|
1828
1951
|
**Key insight**: Design reviews and implementation reviews find
|
|
1829
|
-
**categorically different bugs**. Both phases are necessary.
|
|
1952
|
+
**categorically different bugs**. Both phases are necessary. The corollary that
|
|
1953
|
+
cost eight rounds to learn: **which of the two you are in is readable from the
|
|
1954
|
+
findings' category, and has to be read every round** (§ Step -1 rule 7).
|
|
@@ -82,7 +82,23 @@ module KairosMcp
|
|
|
82
82
|
region = @config['aws_region'] || @config[:aws_region] ||
|
|
83
83
|
ENV.fetch('AWS_REGION', 'us-east-1')
|
|
84
84
|
|
|
85
|
-
Aws::BedrockRuntime::Client.new(region: region)
|
|
85
|
+
Aws::BedrockRuntime::Client.new(region: region, **client_options)
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# The SDK's defaults are a 60 s read timeout and 3 retries, and a retry
|
|
89
|
+
# re-sends the whole prompt. A long generation (a Tier-1 assessment over
|
|
90
|
+
# a full paper) therefore failed after exactly 4 x 60 s with
|
|
91
|
+
# Net::ReadTimeout, having been billed four times (measured 2026-09-12,
|
|
92
|
+
# 243 s). The read timeout follows the same `timeout_seconds` the other
|
|
93
|
+
# adapters honour, and there is no retry: a timed-out generation is not
|
|
94
|
+
# made shorter by sending it again, and the caller decides whether to
|
|
95
|
+
# retry.
|
|
96
|
+
def client_options
|
|
97
|
+
{
|
|
98
|
+
http_open_timeout: 10,
|
|
99
|
+
http_read_timeout: timeout_seconds,
|
|
100
|
+
retry_limit: 0
|
|
101
|
+
}
|
|
86
102
|
end
|
|
87
103
|
|
|
88
104
|
def resolve_model(override)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
# The Bedrock client must carry the configured timeout and no SDK retries.
|
|
3
|
+
# Before 2026-09-12 it was built with the SDK defaults (60 s read timeout,
|
|
4
|
+
# 3 retries): a Tier-1 assessment over a full paper timed out after exactly
|
|
5
|
+
# 4 x 60 s and was billed four times. No AWS gem is needed here: the client
|
|
6
|
+
# constructor is stubbed and its options captured.
|
|
7
|
+
|
|
8
|
+
require 'minitest/autorun'
|
|
9
|
+
require_relative '../lib/llm_client/adapter'
|
|
10
|
+
require_relative '../lib/llm_client/bedrock_adapter'
|
|
11
|
+
|
|
12
|
+
module KairosMcp
|
|
13
|
+
module SkillSets
|
|
14
|
+
module LlmClient
|
|
15
|
+
class TestBedrockAdapterTimeout < Minitest::Test
|
|
16
|
+
def with_stubbed_client
|
|
17
|
+
captured = nil
|
|
18
|
+
fake = Class.new do
|
|
19
|
+
define_singleton_method(:new) { |**opts| captured = opts; :client }
|
|
20
|
+
end
|
|
21
|
+
had_aws = Object.const_defined?(:Aws)
|
|
22
|
+
Object.const_set(:Aws, Module.new) unless had_aws
|
|
23
|
+
had_ns = Aws.const_defined?(:BedrockRuntime)
|
|
24
|
+
Aws.const_set(:BedrockRuntime, Module.new) unless had_ns
|
|
25
|
+
had_client = Aws::BedrockRuntime.const_defined?(:Client)
|
|
26
|
+
orig = had_client ? Aws::BedrockRuntime.send(:remove_const, :Client) : nil
|
|
27
|
+
Aws::BedrockRuntime.const_set(:Client, fake)
|
|
28
|
+
# `require 'aws-sdk-bedrockruntime'` must not fail when the gem is absent.
|
|
29
|
+
$LOADED_FEATURES << 'aws-sdk-bedrockruntime.rb' unless $LOADED_FEATURES.include?('aws-sdk-bedrockruntime.rb')
|
|
30
|
+
yield
|
|
31
|
+
captured
|
|
32
|
+
ensure
|
|
33
|
+
Aws::BedrockRuntime.send(:remove_const, :Client)
|
|
34
|
+
Aws::BedrockRuntime.const_set(:Client, orig) if orig
|
|
35
|
+
Aws.send(:remove_const, :BedrockRuntime) unless had_ns
|
|
36
|
+
Object.send(:remove_const, :Aws) unless had_aws
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def test_configured_timeout_reaches_the_client_and_retries_are_off
|
|
40
|
+
adapter = BedrockAdapter.new({ 'aws_region' => 'eu-central-1', 'timeout_seconds' => 600 })
|
|
41
|
+
opts = with_stubbed_client { adapter.send(:bedrock_client) }
|
|
42
|
+
|
|
43
|
+
assert_equal 'eu-central-1', opts[:region]
|
|
44
|
+
assert_equal 600, opts[:http_read_timeout]
|
|
45
|
+
assert_equal 10, opts[:http_open_timeout]
|
|
46
|
+
assert_equal 0, opts[:retry_limit]
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def test_default_timeout_when_none_configured
|
|
50
|
+
adapter = BedrockAdapter.new({ 'aws_region' => 'eu-central-1' })
|
|
51
|
+
opts = with_stubbed_client { adapter.send(:bedrock_client) }
|
|
52
|
+
assert_equal 120, opts[:http_read_timeout], 'falls back to Adapter#timeout_seconds default'
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|