world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/evals/harbor/ctrf.py
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""Read one harbor trial's CTRF test report into a graded test-pass breakdown.
|
|
2
|
+
|
|
3
|
+
Harbor's pytest verifiers run under `pytest-json-ctrf` and write a CTRF (Common Test Report
|
|
4
|
+
Format) document to `<trial_dir>/verifier/ctrf.json` beside the binary `reward.txt`. That report
|
|
5
|
+
is the only place a trial's per-test outcomes survive, so it is where the resolution a binary
|
|
6
|
+
reward discards comes from (see `wmo.harness.scoring.GradedTests` for why that resolution
|
|
7
|
+
matters and how coarse it is).
|
|
8
|
+
|
|
9
|
+
The shape this parses, confirmed against all 46 reports of the 48-episode TerminalBench-2 probe
|
|
10
|
+
(`pytest 8.4.1` + `pytest-json-ctrf 0.3.5`), which were byte-identical in structure:
|
|
11
|
+
|
|
12
|
+
```json
|
|
13
|
+
{"results": {"tool": {...},
|
|
14
|
+
"summary": {"tests": 2, "passed": 1, "failed": 1, "skipped": 0,
|
|
15
|
+
"pending": 0, "other": 0, "start": ..., "stop": ...},
|
|
16
|
+
"tests": [{"name": "test_outputs.py::test_hello_file_exists", "status": "passed",
|
|
17
|
+
...}, ...]}}
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
`results.summary` is the authoritative aggregate and is read first; `results.tests[]` statuses are
|
|
21
|
+
the fallback when no summary is present, and the tiebreaker when the two disagree (an itemized
|
|
22
|
+
status is a primitive fact, a summary count is derived). Nothing here fabricates a score: a
|
|
23
|
+
missing, unreadable, or empty report yields None, which callers must exclude from graded rates
|
|
24
|
+
rather than average in as 0.0.
|
|
25
|
+
|
|
26
|
+
Multi-step harbor trials relocate their verifier dir to `steps/<step>/verifier/`, so they record no
|
|
27
|
+
graded score today (None, never a zero). Every TerminalBench-2 task WMO distills on is single-step.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import logging
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
|
|
35
|
+
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
|
36
|
+
|
|
37
|
+
from wmo.harness.scoring import GradedTests
|
|
38
|
+
|
|
39
|
+
logger = logging.getLogger(__name__)
|
|
40
|
+
|
|
41
|
+
CTRF_REPORT_FILENAME = "ctrf.json"
|
|
42
|
+
"""Harbor's per-trial CTRF report, written by the verifier into the trial's `verifier/` dir."""
|
|
43
|
+
|
|
44
|
+
_PASSED_STATUS = "passed"
|
|
45
|
+
_FAILED_STATUS = "failed"
|
|
46
|
+
_RESOLVED_STATUSES = frozenset({_PASSED_STATUS, _FAILED_STATUS})
|
|
47
|
+
"""CTRF statuses that carry a verdict. `skipped`, `pending`, and `other` deliberately do not:
|
|
48
|
+
they say the grader never ran the test, which is not the agent's failure and does not stop the
|
|
49
|
+
benchmark's own binary pass either (pytest exits 0 with skips)."""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class _CtrfTest(BaseModel):
|
|
53
|
+
"""One entry of `results.tests[]`; only its status is load-bearing here."""
|
|
54
|
+
|
|
55
|
+
model_config = ConfigDict(extra="ignore")
|
|
56
|
+
|
|
57
|
+
status: str = ""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class _CtrfSummary(BaseModel):
|
|
61
|
+
"""The `results.summary` aggregate; counts are optional so a partial summary degrades."""
|
|
62
|
+
|
|
63
|
+
model_config = ConfigDict(extra="ignore")
|
|
64
|
+
|
|
65
|
+
passed: int | None = Field(default=None, ge=0)
|
|
66
|
+
failed: int | None = Field(default=None, ge=0)
|
|
67
|
+
skipped: int = Field(default=0, ge=0)
|
|
68
|
+
pending: int = Field(default=0, ge=0)
|
|
69
|
+
other: int = Field(default=0, ge=0)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class _CtrfResults(BaseModel):
|
|
73
|
+
"""The `results` object: the summary plus the itemized tests."""
|
|
74
|
+
|
|
75
|
+
model_config = ConfigDict(extra="ignore")
|
|
76
|
+
|
|
77
|
+
summary: _CtrfSummary | None = None
|
|
78
|
+
tests: list[_CtrfTest] = Field(default_factory=list)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class _CtrfReport(BaseModel):
|
|
82
|
+
"""The CTRF document root."""
|
|
83
|
+
|
|
84
|
+
model_config = ConfigDict(extra="ignore")
|
|
85
|
+
|
|
86
|
+
results: _CtrfResults
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def read_trial_graded_tests(trial_dir: Path) -> GradedTests | None:
|
|
90
|
+
"""The per-test breakdown one harbor trial's verifier recorded, when it recorded one.
|
|
91
|
+
|
|
92
|
+
Args:
|
|
93
|
+
trial_dir: The harbor trial directory (a `ScoreCell.artifact_dir`).
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
The trial's `GradedTests`, or None when no graded score exists for it: no report file,
|
|
97
|
+
an unreadable or malformed one, or a report in which no test returned a verdict. None is
|
|
98
|
+
never a 0.0 and callers must keep it out of graded denominators rather than count it as a
|
|
99
|
+
failure, the same rule that keeps an ungradeable trial out of `solve_rate`.
|
|
100
|
+
"""
|
|
101
|
+
path = trial_dir / "verifier" / CTRF_REPORT_FILENAME
|
|
102
|
+
try:
|
|
103
|
+
text = path.read_text(encoding="utf-8")
|
|
104
|
+
except (OSError, UnicodeDecodeError):
|
|
105
|
+
# Overwhelmingly the ordinary case (no report was written), so this is not a warning: the
|
|
106
|
+
# verifier-failure diagnosis belongs to the scorer's `infra_failed` note.
|
|
107
|
+
logger.debug("no readable CTRF report at %s; this trial records no graded score", path)
|
|
108
|
+
return None
|
|
109
|
+
try:
|
|
110
|
+
report = _CtrfReport.model_validate_json(text)
|
|
111
|
+
except ValidationError:
|
|
112
|
+
logger.warning(
|
|
113
|
+
"CTRF report %s does not parse as a CTRF document; this trial records no graded "
|
|
114
|
+
"score (its binary reward is unaffected)",
|
|
115
|
+
path,
|
|
116
|
+
exc_info=True,
|
|
117
|
+
)
|
|
118
|
+
return None
|
|
119
|
+
return _breakdown(report.results, path)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _breakdown(results: _CtrfResults, path: Path) -> GradedTests | None:
|
|
123
|
+
"""One report's counts, preferring itemized statuses when they contradict the summary."""
|
|
124
|
+
from_summary = _from_summary(results.summary)
|
|
125
|
+
from_tests = _from_tests(results.tests)
|
|
126
|
+
if from_summary is not None and from_tests is not None and from_summary != from_tests:
|
|
127
|
+
logger.warning(
|
|
128
|
+
"CTRF report %s disagrees with itself: summary says %s, its %d itemized test(s) say "
|
|
129
|
+
"%s; using the itemized statuses, which are the primitive fact",
|
|
130
|
+
path,
|
|
131
|
+
from_summary,
|
|
132
|
+
len(results.tests),
|
|
133
|
+
from_tests,
|
|
134
|
+
)
|
|
135
|
+
return from_tests
|
|
136
|
+
if from_summary is not None:
|
|
137
|
+
return from_summary
|
|
138
|
+
if from_tests is None:
|
|
139
|
+
logger.warning(
|
|
140
|
+
"CTRF report %s carries no test that returned a verdict (no summary counts and no "
|
|
141
|
+
"passed/failed entries); this trial records no graded score",
|
|
142
|
+
path,
|
|
143
|
+
)
|
|
144
|
+
return from_tests
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _from_summary(summary: _CtrfSummary | None) -> GradedTests | None:
|
|
148
|
+
"""The breakdown `results.summary` states, or None when it states no verdict counts."""
|
|
149
|
+
if summary is None or summary.passed is None or summary.failed is None:
|
|
150
|
+
return None
|
|
151
|
+
resolved = summary.passed + summary.failed
|
|
152
|
+
if resolved == 0:
|
|
153
|
+
return None
|
|
154
|
+
return GradedTests(
|
|
155
|
+
passed=summary.passed,
|
|
156
|
+
resolved=resolved,
|
|
157
|
+
unresolved=summary.skipped + summary.pending + summary.other,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _from_tests(tests: list[_CtrfTest]) -> GradedTests | None:
|
|
162
|
+
"""The breakdown the itemized `results.tests[]` statuses state, or None when none has a verdict.
|
|
163
|
+
|
|
164
|
+
An unrecognized status counts as unresolved rather than as a failure: it means this parser does
|
|
165
|
+
not know what the grader said, which is not evidence against the agent.
|
|
166
|
+
"""
|
|
167
|
+
passed = sum(1 for test in tests if test.status == _PASSED_STATUS)
|
|
168
|
+
resolved = sum(1 for test in tests if test.status in _RESOLVED_STATUSES)
|
|
169
|
+
if resolved == 0:
|
|
170
|
+
return None
|
|
171
|
+
return GradedTests(passed=passed, resolved=resolved, unresolved=len(tests) - resolved)
|