okstra 0.169.0 → 0.169.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/cli.md +1 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/bin/okstra-error-log.py +38 -282
- package/runtime/prompts/lead/convergence.md +53 -7
- package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
- package/runtime/python/okstra_ctl/agent_invocation.py +86 -6
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +8 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +226 -6
- package/runtime/python/okstra_ctl/error_log_write.py +308 -0
- package/runtime/python/okstra_ctl/worker_audit_check.py +26 -4
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +59 -9
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +24 -1
- package/runtime/python/okstra_ctl/worker_prompt_headers.py +2 -2
|
@@ -45,6 +45,7 @@ from .final_report_paths import (
|
|
|
45
45
|
final_report_data_path as _final_report_data_path,
|
|
46
46
|
final_report_markdown_path as _final_report_markdown_path,
|
|
47
47
|
)
|
|
48
|
+
from .error_log_write import append_observed
|
|
48
49
|
from .lead_events import LeadEvent, append_lead_event
|
|
49
50
|
from .initial_prompt_materialization import (
|
|
50
51
|
InitialPromptMaterializationError,
|
|
@@ -60,13 +61,27 @@ from .report_finalize import (
|
|
|
60
61
|
FinalizeError,
|
|
61
62
|
run_finalize,
|
|
62
63
|
)
|
|
64
|
+
from .worker_audit_ledger import (
|
|
65
|
+
check_worker_results_audit,
|
|
66
|
+
parse_worker_result_name,
|
|
67
|
+
)
|
|
63
68
|
from .worker_prompt_body import REPORT_WRITER_WORKER_ID
|
|
69
|
+
from .worker_prompt_headers import (
|
|
70
|
+
WorkerPromptHeaderError,
|
|
71
|
+
resolve_errors_log_path,
|
|
72
|
+
)
|
|
64
73
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
65
74
|
from .wrapper_status import read_wrapper_status, status_path_for_prompt
|
|
66
75
|
|
|
67
76
|
|
|
68
77
|
MAX_WORKER_ATTEMPTS = 2
|
|
69
78
|
TERMINAL_DISPATCH_STATUSES = {"completed", "timeout", "error", "not-run"}
|
|
79
|
+
# What the error log records for a wrapper the dispatcher timed out, matching
|
|
80
|
+
# the value `team-contract` prescribes for a polling-cap termination.
|
|
81
|
+
_WRAPPER_TIMEOUT_EXIT_CODE = 124
|
|
82
|
+
# The excerpt shares one atomic PIPE_BUF append with the rest of the record, so
|
|
83
|
+
# it is capped far below the writer's own 2048-byte stderr limit.
|
|
84
|
+
_WRAPPER_LOG_TAIL_BYTES = 800
|
|
70
85
|
|
|
71
86
|
|
|
72
87
|
@dataclass(frozen=True)
|
|
@@ -927,13 +942,22 @@ def _retry_from_record(
|
|
|
927
942
|
worker_id = _require_string(record, "workerId")
|
|
928
943
|
attempt = int(record.get("attempt", 1))
|
|
929
944
|
job = _job_from_record(plan.project_root, record)
|
|
930
|
-
|
|
931
|
-
|
|
945
|
+
reason = "required worker artifact was not produced"
|
|
946
|
+
_update_dispatch_status(plan.team_state_path, job, attempt, "error", reason)
|
|
947
|
+
# `team-contract` counts the first attempt's failure as a recorded
|
|
948
|
+
# `cli-failure`; a retry that succeeds settles `completed` and would
|
|
949
|
+
# otherwise leave no trace that anything had to be re-run.
|
|
950
|
+
details: dict[str, Any] = {"workerId": worker_id, "attempt": attempt}
|
|
951
|
+
details["errorLogAppend"] = _record_wrapper_failure(
|
|
952
|
+
plan, job, attempt, outcome, reason
|
|
953
|
+
)
|
|
954
|
+
_append_event(plan, "worker-retry-scheduled", details)
|
|
932
955
|
_spawn_job(plan, job, attempt + 1)
|
|
933
956
|
|
|
934
957
|
|
|
935
958
|
def _finish_attempt(plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: WorkerOutcome) -> None:
|
|
936
|
-
|
|
959
|
+
settlement = _settle(plan, job, attempt, outcome)
|
|
960
|
+
if settlement.completed:
|
|
937
961
|
if job.invocation_id:
|
|
938
962
|
_link_agent_dispatch_result(
|
|
939
963
|
project_root=plan.project_root,
|
|
@@ -965,16 +989,23 @@ def _finish_attempt(plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: W
|
|
|
965
989
|
_transition_worker_status(
|
|
966
990
|
plan.team_state_path, job.worker_id, "completed", ""
|
|
967
991
|
)
|
|
968
|
-
_update_dispatch_status(
|
|
992
|
+
_update_dispatch_status(
|
|
993
|
+
plan.team_state_path, job, attempt, "completed", settlement.note
|
|
994
|
+
)
|
|
969
995
|
details = _result_details(job, attempt, outcome)
|
|
970
996
|
details["postProcessing"] = post_process["steps"]
|
|
997
|
+
if settlement.error_log_append is not None:
|
|
998
|
+
details["errorLogAppend"] = settlement.error_log_append
|
|
971
999
|
_append_event(plan, "worker-result-collected", details)
|
|
972
1000
|
return
|
|
973
|
-
reason =
|
|
1001
|
+
reason = settlement.reason
|
|
974
1002
|
status = "timeout" if outcome.timeout else "error"
|
|
975
1003
|
_transition_worker_status(plan.team_state_path, job.worker_id, status, reason)
|
|
976
1004
|
_update_dispatch_status(plan.team_state_path, job, attempt, status, reason)
|
|
977
|
-
|
|
1005
|
+
details = _failure_details(job, attempt, outcome, reason)
|
|
1006
|
+
if settlement.error_log_append is not None:
|
|
1007
|
+
details["errorLogAppend"] = settlement.error_log_append
|
|
1008
|
+
_append_event(plan, "worker-failed", details)
|
|
978
1009
|
|
|
979
1010
|
|
|
980
1011
|
def _finish_record(plan: DispatchPlan, record: Mapping[str, Any], outcome: WorkerOutcome) -> None:
|
|
@@ -982,6 +1013,191 @@ def _finish_record(plan: DispatchPlan, record: Mapping[str, Any], outcome: Worke
|
|
|
982
1013
|
_finish_attempt(plan, job, int(record.get("attempt", 1)), outcome)
|
|
983
1014
|
|
|
984
1015
|
|
|
1016
|
+
@dataclass(frozen=True)
|
|
1017
|
+
class _Settlement:
|
|
1018
|
+
"""How one attempt's terminal status was decided."""
|
|
1019
|
+
|
|
1020
|
+
completed: bool
|
|
1021
|
+
reason: str
|
|
1022
|
+
note: str
|
|
1023
|
+
error_log_append: dict[str, Any] | None
|
|
1024
|
+
|
|
1025
|
+
|
|
1026
|
+
def _settle(
|
|
1027
|
+
plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: WorkerOutcome
|
|
1028
|
+
) -> _Settlement:
|
|
1029
|
+
"""Judge an attempt by its artifacts, not by the wrapper's exit code alone.
|
|
1030
|
+
|
|
1031
|
+
A wrapper can die after its worker has already written everything — an
|
|
1032
|
+
observed case is a connection dropped at session teardown, long after the
|
|
1033
|
+
result file and its audit sidecar were on disk. Settling that as `error`
|
|
1034
|
+
discards a complete analysis, and not figuratively: `convergence_engine`
|
|
1035
|
+
admits only dispatches that settled `completed`, so the worker's findings
|
|
1036
|
+
never reach re-verification. The lead's own re-dispatch triggers agree —
|
|
1037
|
+
they name a missing, unparseable, or audit-failing result, never an exit
|
|
1038
|
+
code — but the only signal `team await` gave was the status.
|
|
1039
|
+
|
|
1040
|
+
So a non-zero exit with every completion path present is re-judged by the
|
|
1041
|
+
audit-sidecar contract, the same rules `okstra worker-audit-check` runs. It
|
|
1042
|
+
passes and the dispatch settles `completed`; it fails and the dispatch stays
|
|
1043
|
+
`error` exactly as before. Either way the wrapper's failure is written to the
|
|
1044
|
+
run error log, so a `completed` here is never a swallowed failure.
|
|
1045
|
+
"""
|
|
1046
|
+
if outcome.returncode == 0 and not outcome.missing_completion_paths and not outcome.timeout:
|
|
1047
|
+
return _Settlement(True, "", "", None)
|
|
1048
|
+
reason = _failure_reason(outcome)
|
|
1049
|
+
completed = False
|
|
1050
|
+
note = ""
|
|
1051
|
+
if not outcome.timeout and not outcome.missing_completion_paths:
|
|
1052
|
+
audit_failures = _audit_sidecar_failures(job)
|
|
1053
|
+
if audit_failures:
|
|
1054
|
+
reason = (
|
|
1055
|
+
f"{reason}; worker artifacts failed the audit-sidecar "
|
|
1056
|
+
f"contract: {audit_failures[0]}"
|
|
1057
|
+
)
|
|
1058
|
+
else:
|
|
1059
|
+
completed = True
|
|
1060
|
+
note = (
|
|
1061
|
+
f"{reason}, but every completion artifact was written and "
|
|
1062
|
+
f"passed the audit-sidecar contract"
|
|
1063
|
+
)
|
|
1064
|
+
append = _record_wrapper_failure(plan, job, attempt, outcome, note or reason)
|
|
1065
|
+
return _Settlement(completed, "" if completed else reason, note, append)
|
|
1066
|
+
|
|
1067
|
+
|
|
1068
|
+
def _audit_sidecar_failures(job: WorkerJob) -> tuple[str, ...]:
|
|
1069
|
+
"""This worker's audit-sidecar contract failures, if the check can run.
|
|
1070
|
+
|
|
1071
|
+
The check's arguments come from the result filename rather than the manifest
|
|
1072
|
+
so the scan cannot widen past the file this job produced: `worker-results/`
|
|
1073
|
+
accumulates every run's artifacts, and the `worker=` filter matches the
|
|
1074
|
+
`-worker`-suffixed role, not the bare provider id. A non-canonical name
|
|
1075
|
+
leaves nothing to enforce, and an unverifiable artifact must not be promoted
|
|
1076
|
+
to `completed`, so that reports one failure rather than an empty tuple.
|
|
1077
|
+
"""
|
|
1078
|
+
parsed = parse_worker_result_name(job.worker_result_path.name)
|
|
1079
|
+
if parsed is None:
|
|
1080
|
+
return (
|
|
1081
|
+
f"worker result `{job.worker_result_path.name}` is not a canonical "
|
|
1082
|
+
f"`<role>-worker-<task-type>-<seq>.md` name, so the audit-sidecar "
|
|
1083
|
+
f"contract could not be checked",
|
|
1084
|
+
)
|
|
1085
|
+
return tuple(
|
|
1086
|
+
check_worker_results_audit(
|
|
1087
|
+
job.worker_result_path.parent.parent,
|
|
1088
|
+
parsed.task_type,
|
|
1089
|
+
parsed.seq,
|
|
1090
|
+
worker=parsed.worker_role,
|
|
1091
|
+
)
|
|
1092
|
+
)
|
|
1093
|
+
|
|
1094
|
+
|
|
1095
|
+
def _record_wrapper_failure(
|
|
1096
|
+
plan: DispatchPlan,
|
|
1097
|
+
job: WorkerJob,
|
|
1098
|
+
attempt: int,
|
|
1099
|
+
outcome: WorkerOutcome,
|
|
1100
|
+
message: str,
|
|
1101
|
+
) -> dict[str, Any]:
|
|
1102
|
+
"""Write the wrapper's own failure to the run-level error log.
|
|
1103
|
+
|
|
1104
|
+
`okstra-lead-contract` tells Lead the deterministic dispatcher records this
|
|
1105
|
+
and that Lead does not need to re-record it. Nothing did: no code path
|
|
1106
|
+
anywhere called the error-log writer, so every wrapper failure vanished, and
|
|
1107
|
+
with it the `instruction-set/prior-run-errors.md` digest the next run reads
|
|
1108
|
+
and the `/okstra-inspect errors` report. The dispatcher is the only component
|
|
1109
|
+
that holds the exit code, so it is the one that writes.
|
|
1110
|
+
|
|
1111
|
+
Never raises. Logging is bookkeeping around a dispatch that has already
|
|
1112
|
+
settled; letting a rejected or unwritable record throw here would turn a
|
|
1113
|
+
recorded outcome into an unrecorded crash. What went wrong travels back in
|
|
1114
|
+
the lead event instead.
|
|
1115
|
+
"""
|
|
1116
|
+
result: dict[str, Any] = {"ok": False, "reason": "", "path": ""}
|
|
1117
|
+
try:
|
|
1118
|
+
out_path = resolve_errors_log_path(
|
|
1119
|
+
plan.project_root,
|
|
1120
|
+
plan.manifest,
|
|
1121
|
+
_load_optional_json(
|
|
1122
|
+
plan.project_root, plan.manifest.get("activeRunContextPath")
|
|
1123
|
+
),
|
|
1124
|
+
)
|
|
1125
|
+
result["path"] = str(out_path)
|
|
1126
|
+
append_observed(
|
|
1127
|
+
out_path=out_path,
|
|
1128
|
+
task_key=_string_value(plan.manifest.get("taskKey")),
|
|
1129
|
+
phase=_workflow_phase(plan.manifest),
|
|
1130
|
+
agent=_error_log_agent(job.worker_id),
|
|
1131
|
+
agent_role=(
|
|
1132
|
+
"report-writer"
|
|
1133
|
+
if job.worker_id == REPORT_WRITER_WORKER_ID
|
|
1134
|
+
else "worker"
|
|
1135
|
+
),
|
|
1136
|
+
model=job.model_execution_value,
|
|
1137
|
+
error_type="cli-failure",
|
|
1138
|
+
command=" ".join(job.command),
|
|
1139
|
+
command_kind="wrapper",
|
|
1140
|
+
exit_code=_WRAPPER_TIMEOUT_EXIT_CODE if outcome.timeout else outcome.returncode,
|
|
1141
|
+
duration_ms=_wrapper_duration_ms(outcome),
|
|
1142
|
+
message=f"attempt {attempt}: {message}",
|
|
1143
|
+
stderr_excerpt=_wrapper_log_tail(job),
|
|
1144
|
+
context=None,
|
|
1145
|
+
)
|
|
1146
|
+
except (OSError, ValueError, TypeError, WorkerPromptHeaderError) as exc:
|
|
1147
|
+
result["reason"] = f"{type(exc).__name__}: {exc}"
|
|
1148
|
+
return result
|
|
1149
|
+
result["ok"] = True
|
|
1150
|
+
return result
|
|
1151
|
+
|
|
1152
|
+
|
|
1153
|
+
def _error_log_agent(worker_id: str) -> str:
|
|
1154
|
+
"""The error log's `--agent` enum value for a worker id.
|
|
1155
|
+
|
|
1156
|
+
The log's own allow-list is the authority on what it accepts; a worker whose
|
|
1157
|
+
name is outside it is reported as such by `append_observed` rather than
|
|
1158
|
+
silently rewritten into some other agent's records.
|
|
1159
|
+
"""
|
|
1160
|
+
if worker_id == REPORT_WRITER_WORKER_ID:
|
|
1161
|
+
return REPORT_WRITER_WORKER_ID
|
|
1162
|
+
return f"{worker_id}-worker"
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
def _workflow_phase(manifest: Mapping[str, Any]) -> str:
|
|
1166
|
+
workflow = manifest.get("workflow")
|
|
1167
|
+
if isinstance(workflow, Mapping):
|
|
1168
|
+
return _string_value(workflow.get("currentPhase"))
|
|
1169
|
+
return ""
|
|
1170
|
+
|
|
1171
|
+
|
|
1172
|
+
def _wrapper_duration_ms(outcome: WorkerOutcome) -> int | None:
|
|
1173
|
+
if outcome.status_sidecar_path is None:
|
|
1174
|
+
return None
|
|
1175
|
+
status = read_wrapper_status(outcome.status_sidecar_path)
|
|
1176
|
+
if status is None:
|
|
1177
|
+
return None
|
|
1178
|
+
value = status.raw.get("duration_ms")
|
|
1179
|
+
return value if isinstance(value, int) and not isinstance(value, bool) else None
|
|
1180
|
+
|
|
1181
|
+
|
|
1182
|
+
def _wrapper_log_tail(job: WorkerJob) -> str | None:
|
|
1183
|
+
"""The tail of the wrapper transcript, which usually names the real failure.
|
|
1184
|
+
|
|
1185
|
+
The observed case put `API Error: Connection lost mid-response.` in the last
|
|
1186
|
+
two lines and nothing anywhere else; without it the record says only that
|
|
1187
|
+
some process exited 1. Capped well under the writer's own excerpt limit
|
|
1188
|
+
because a whole record must stay inside one atomic `PIPE_BUF` append.
|
|
1189
|
+
"""
|
|
1190
|
+
log_path = job.prompt_path.with_suffix(job.prompt_path.suffix + ".log")
|
|
1191
|
+
try:
|
|
1192
|
+
with log_path.open("rb") as handle:
|
|
1193
|
+
handle.seek(0, 2)
|
|
1194
|
+
handle.seek(max(0, handle.tell() - _WRAPPER_LOG_TAIL_BYTES))
|
|
1195
|
+
tail = handle.read()
|
|
1196
|
+
except OSError:
|
|
1197
|
+
return None
|
|
1198
|
+
return tail.decode("utf-8", errors="replace").strip() or None
|
|
1199
|
+
|
|
1200
|
+
|
|
985
1201
|
def _post_process_report_writer_result(
|
|
986
1202
|
plan: DispatchPlan,
|
|
987
1203
|
job: WorkerJob,
|
|
@@ -1262,6 +1478,10 @@ def _result_details(job: WorkerJob, attempt: int, outcome: WorkerOutcome) -> dic
|
|
|
1262
1478
|
"attempt": attempt,
|
|
1263
1479
|
"dispatchMode": BACKEND_CLI_WRAPPER if outcome.degraded_from else job.backend,
|
|
1264
1480
|
"missingCompletionPaths": [],
|
|
1481
|
+
# A collected result can still come from a wrapper that exited non-zero
|
|
1482
|
+
# (`_settle`). The status says the artifacts are good; this says what the
|
|
1483
|
+
# process did, so the trace never loses one fact to the other.
|
|
1484
|
+
"wrapperExitCode": outcome.returncode,
|
|
1265
1485
|
}
|
|
1266
1486
|
|
|
1267
1487
|
|
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
"""Writer core for runs/<task-type>/logs/errors-<task-type>-<seq>.jsonl.
|
|
2
|
+
|
|
3
|
+
Every error record in a run log is appended through this module. It lives here
|
|
4
|
+
rather than inside `scripts/okstra-error-log.py` because the CLI is no longer the
|
|
5
|
+
only caller: the deterministic dispatcher records a worker wrapper's non-zero
|
|
6
|
+
exit as a `cli-failure` in-process (`dispatch_core`), and the contract sentence
|
|
7
|
+
that says it does is only true while both paths share one writer. The script
|
|
8
|
+
keeps the argparse surface and re-exports these names.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import datetime as dt
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from .models import provider_ids
|
|
18
|
+
|
|
19
|
+
STDERR_EXCERPT_MAX_BYTES = 2048
|
|
20
|
+
TRUNCATION_SUFFIX = "...[truncated]"
|
|
21
|
+
PIPE_BUF_BYTES = 4096
|
|
22
|
+
|
|
23
|
+
ALLOWED_ERROR_TYPES = {"tool-failure", "cli-failure", "contract-violation"}
|
|
24
|
+
# Derived, not listed. The hand-written set had drifted two providers behind the
|
|
25
|
+
# registry: `grok` and `kimi` ship worker definitions and can be dispatched, but
|
|
26
|
+
# their agent names were absent, so every error they reported was rejected at
|
|
27
|
+
# the argument parser — silently, for anyone who did not read the exit code.
|
|
28
|
+
# The registry is where a provider is added, so it is where this follows from.
|
|
29
|
+
ALLOWED_AGENTS = (
|
|
30
|
+
{f"{provider}-worker" for provider in provider_ids("analyser")}
|
|
31
|
+
# The report writer is not an analyser and is named without the suffix,
|
|
32
|
+
# matching `REPORT_WRITER_WORKER_ID`.
|
|
33
|
+
| {"report-writer"}
|
|
34
|
+
# Lead identities come from the selected host adapter rather than the
|
|
35
|
+
# provider registry; `adapters/hosts/*/relay.md` names the value to pass.
|
|
36
|
+
| {"claude-lead"}
|
|
37
|
+
)
|
|
38
|
+
ALLOWED_AGENT_ROLES = {"lead", "worker", "report-writer"}
|
|
39
|
+
SUPPORTED_SIDECAR_SCHEMA_VERSIONS = {1}
|
|
40
|
+
|
|
41
|
+
ALLOWED_CAUSES = {
|
|
42
|
+
"sandbox-denied", "service-unavailable", "auth-failed", "unknown",
|
|
43
|
+
}
|
|
44
|
+
# A `sandbox-denied` claim is only admissible with both probes attached.
|
|
45
|
+
CAUSE_EVIDENCE_FIELDS = ("targetProbe", "controlProbe")
|
|
46
|
+
# Both probes share a record's PIPE_BUF_BYTES budget with stderrExcerpt, so
|
|
47
|
+
# they cannot reuse the 2048 cap that assumes stderrExcerpt owns it alone.
|
|
48
|
+
CAUSE_PROBE_MAX_BYTES = 256
|
|
49
|
+
# Backstop vocabulary: scanned in `message` only — never in stderrExcerpt,
|
|
50
|
+
# where a kernel's real "Operation not permitted" is legitimate content.
|
|
51
|
+
# Deliberately excludes bare "blocked"/"blocks": everyday English that would
|
|
52
|
+
# reject honest records like "test blocked on upstream dependency".
|
|
53
|
+
_BLOCKING_CLAIM_TERMS = (
|
|
54
|
+
"sandbox", "not permitted", "permission denied", "eperm",
|
|
55
|
+
)
|
|
56
|
+
# The backstop targets *unclassified* blocking claims. A worker that declared
|
|
57
|
+
# a specific cause has already done the honest work — `auth-failed` legitimately
|
|
58
|
+
# reads "permission denied" (MySQL 1045).
|
|
59
|
+
_UNCLASSIFIED_CAUSES = (None, "unknown")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _now_utc():
|
|
63
|
+
return dt.datetime.now(dt.timezone.utc)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _iso(t):
|
|
67
|
+
return t.isoformat()
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _truncate_utf8(s, limit):
|
|
71
|
+
"""Truncate to `limit` bytes without splitting a multibyte character."""
|
|
72
|
+
if s is None:
|
|
73
|
+
return None
|
|
74
|
+
encoded = s.encode("utf-8")
|
|
75
|
+
if len(encoded) <= limit:
|
|
76
|
+
return s
|
|
77
|
+
cut = encoded[:limit]
|
|
78
|
+
while cut:
|
|
79
|
+
try:
|
|
80
|
+
return cut.decode("utf-8") + TRUNCATION_SUFFIX
|
|
81
|
+
except UnicodeDecodeError:
|
|
82
|
+
cut = cut[:-1]
|
|
83
|
+
return TRUNCATION_SUFFIX
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def truncate_stderr(s):
|
|
87
|
+
"""Truncate stderr text to STDERR_EXCERPT_MAX_BYTES, multibyte-safe."""
|
|
88
|
+
return _truncate_utf8(s, STDERR_EXCERPT_MAX_BYTES)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def normalize_cause_context(context, *, message):
|
|
92
|
+
"""Validate a record's cause claim and return the normalized context.
|
|
93
|
+
|
|
94
|
+
Raises ValueError on three conditions:
|
|
95
|
+
- `cause` is set to a value outside ALLOWED_CAUSES;
|
|
96
|
+
- `cause` is 'sandbox-denied' but the two probes are missing or blank —
|
|
97
|
+
those probes are what distinguish a real denial from an unreachable
|
|
98
|
+
or auth-gated target, the misdiagnosis this gate exists to stop;
|
|
99
|
+
- `message` asserts a block in prose while the record left its cause
|
|
100
|
+
unclassified, which would smuggle the same claim past the gate.
|
|
101
|
+
"""
|
|
102
|
+
cause = context.get("cause") if isinstance(context, dict) else None
|
|
103
|
+
|
|
104
|
+
if cause is not None and cause not in ALLOWED_CAUSES:
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"invalid cause: {cause!r} (allowed: {sorted(ALLOWED_CAUSES)})"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
if cause == "sandbox-denied":
|
|
110
|
+
evidence = context.get("causeEvidence")
|
|
111
|
+
if not isinstance(evidence, dict):
|
|
112
|
+
raise ValueError(
|
|
113
|
+
"cause 'sandbox-denied' requires context.causeEvidence with "
|
|
114
|
+
f"{list(CAUSE_EVIDENCE_FIELDS)}"
|
|
115
|
+
)
|
|
116
|
+
normalized_evidence = {}
|
|
117
|
+
for field in CAUSE_EVIDENCE_FIELDS:
|
|
118
|
+
value = evidence.get(field)
|
|
119
|
+
if not isinstance(value, str) or not value.strip():
|
|
120
|
+
raise ValueError(
|
|
121
|
+
f"cause 'sandbox-denied' requires a non-empty "
|
|
122
|
+
f"context.causeEvidence.{field}: record the command and "
|
|
123
|
+
f"its raw output that proves the claim"
|
|
124
|
+
)
|
|
125
|
+
normalized_evidence[field] = _truncate_utf8(
|
|
126
|
+
value, CAUSE_PROBE_MAX_BYTES
|
|
127
|
+
)
|
|
128
|
+
return {**context, "causeEvidence": normalized_evidence}
|
|
129
|
+
|
|
130
|
+
if message and cause in _UNCLASSIFIED_CAUSES:
|
|
131
|
+
lowered = message.lower()
|
|
132
|
+
hit = next((t for t in _BLOCKING_CLAIM_TERMS if t in lowered), None)
|
|
133
|
+
if hit:
|
|
134
|
+
raise ValueError(
|
|
135
|
+
f"message asserts a blocking claim ({hit!r}) without "
|
|
136
|
+
"context.cause='sandbox-denied' + context.causeEvidence. "
|
|
137
|
+
"Either attach the two probes, or state the cause you "
|
|
138
|
+
"actually verified."
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
return context
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def append_jsonl_line(path, record):
|
|
145
|
+
"""Append a single JSON record as one line to ``path``.
|
|
146
|
+
|
|
147
|
+
Atomicity guarantee (POSIX only):
|
|
148
|
+
With ``O_APPEND`` and a single ``write()`` syscall, the kernel
|
|
149
|
+
appends the entire payload as one indivisible operation as long as
|
|
150
|
+
the payload size is at most ``PIPE_BUF`` (4096 bytes on Linux and
|
|
151
|
+
macOS). Larger payloads may be split across syscalls and interleave
|
|
152
|
+
with concurrent writers, so this helper rejects them with
|
|
153
|
+
``ValueError`` rather than silently losing atomicity.
|
|
154
|
+
|
|
155
|
+
The atomicity contract holds only on POSIX filesystems with O_APPEND
|
|
156
|
+
semantics. Concurrent writers using ``O_TRUNC``, ``unlink``, or
|
|
157
|
+
non-append modes against the same path break the contract and are
|
|
158
|
+
out of scope for this helper.
|
|
159
|
+
|
|
160
|
+
Caller responsibilities:
|
|
161
|
+
- Keep records small (this module's stderr excerpt cap of
|
|
162
|
+
``STDERR_EXCERPT_MAX_BYTES`` exists to keep records well under
|
|
163
|
+
``PIPE_BUF_BYTES``).
|
|
164
|
+
- Handle ``TypeError`` from ``json.dumps`` for non-serializable values.
|
|
165
|
+
|
|
166
|
+
Creates parent directories as needed.
|
|
167
|
+
"""
|
|
168
|
+
p = Path(path)
|
|
169
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
170
|
+
# ensure_ascii=False keeps UTF-8 compact (no \uXXXX escapes).
|
|
171
|
+
# json.dumps escapes literal newlines inside string values, so the
|
|
172
|
+
# only unescaped newline is the record separator we append below.
|
|
173
|
+
line = json.dumps(record, ensure_ascii=False, separators=(",", ":")) + "\n"
|
|
174
|
+
data = line.encode("utf-8")
|
|
175
|
+
if len(data) > PIPE_BUF_BYTES:
|
|
176
|
+
raise ValueError(
|
|
177
|
+
f"record too large for atomic append: {len(data)} bytes > "
|
|
178
|
+
f"PIPE_BUF ({PIPE_BUF_BYTES})"
|
|
179
|
+
)
|
|
180
|
+
# mode 0o644: owner read/write, group/world read-only.
|
|
181
|
+
fd = os.open(str(p), os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o644)
|
|
182
|
+
try:
|
|
183
|
+
os.write(fd, data)
|
|
184
|
+
finally:
|
|
185
|
+
os.close(fd)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def append_observed(
|
|
189
|
+
*,
|
|
190
|
+
out_path,
|
|
191
|
+
task_key,
|
|
192
|
+
phase,
|
|
193
|
+
agent,
|
|
194
|
+
agent_role,
|
|
195
|
+
model,
|
|
196
|
+
error_type,
|
|
197
|
+
command,
|
|
198
|
+
command_kind,
|
|
199
|
+
exit_code,
|
|
200
|
+
duration_ms,
|
|
201
|
+
message,
|
|
202
|
+
stderr_excerpt,
|
|
203
|
+
context,
|
|
204
|
+
now=None,
|
|
205
|
+
):
|
|
206
|
+
"""Append a lead-observed error event to errors.jsonl."""
|
|
207
|
+
if error_type not in ALLOWED_ERROR_TYPES:
|
|
208
|
+
raise ValueError(f"invalid errorType: {error_type!r}")
|
|
209
|
+
if agent not in ALLOWED_AGENTS:
|
|
210
|
+
raise ValueError(f"invalid agent: {agent!r}")
|
|
211
|
+
if agent_role not in ALLOWED_AGENT_ROLES:
|
|
212
|
+
raise ValueError(f"invalid agentRole: {agent_role!r}")
|
|
213
|
+
# Runs before append_jsonl_line so a rejected claim leaves no trace in the
|
|
214
|
+
# log: a written-then-flagged record is still a record someone can cite.
|
|
215
|
+
context = normalize_cause_context(context, message=message)
|
|
216
|
+
ts = _iso(now or _now_utc())
|
|
217
|
+
rec = {
|
|
218
|
+
"ts": ts,
|
|
219
|
+
"recordedAt": ts,
|
|
220
|
+
"taskKey": task_key,
|
|
221
|
+
"phase": str(phase),
|
|
222
|
+
"agent": agent,
|
|
223
|
+
"agentRole": agent_role,
|
|
224
|
+
"model": model,
|
|
225
|
+
"source": "lead-observed",
|
|
226
|
+
"errorType": error_type,
|
|
227
|
+
"command": command,
|
|
228
|
+
"commandKind": command_kind,
|
|
229
|
+
"exitCode": exit_code,
|
|
230
|
+
"durationMs": duration_ms,
|
|
231
|
+
"message": message,
|
|
232
|
+
"stderrExcerpt": truncate_stderr(stderr_excerpt),
|
|
233
|
+
"context": context,
|
|
234
|
+
}
|
|
235
|
+
append_jsonl_line(out_path, rec)
|
|
236
|
+
return rec
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def dump_from_worker_sidecar(
|
|
240
|
+
*,
|
|
241
|
+
sidecar_path,
|
|
242
|
+
out_path,
|
|
243
|
+
task_key,
|
|
244
|
+
agent,
|
|
245
|
+
agent_role,
|
|
246
|
+
model,
|
|
247
|
+
now=None,
|
|
248
|
+
):
|
|
249
|
+
"""Read worker sidecar errors[] and append each to errors.jsonl with
|
|
250
|
+
Lead-side metadata filled in. Returns number of records appended.
|
|
251
|
+
|
|
252
|
+
Raises ValueError if:
|
|
253
|
+
- ``agent`` or ``agent_role`` is not in the allow-lists
|
|
254
|
+
- sidecar ``schemaVersion`` is not in ``SUPPORTED_SIDECAR_SCHEMA_VERSIONS``
|
|
255
|
+
- any entry's ``errorType`` is not in ``ALLOWED_ERROR_TYPES``
|
|
256
|
+
- any entry asserts a blocking cause without its required evidence
|
|
257
|
+
|
|
258
|
+
Returns 0 (no-op) if the sidecar file does not exist or its
|
|
259
|
+
``errors`` list is empty.
|
|
260
|
+
|
|
261
|
+
Partial-failure semantics: entries are validated and appended in
|
|
262
|
+
order. If entry N fails validation, entries 0..N-1 have already
|
|
263
|
+
been written to ``out_path`` and are NOT rolled back. Callers that
|
|
264
|
+
require atomicity must validate the sidecar payload before invoking
|
|
265
|
+
this function.
|
|
266
|
+
"""
|
|
267
|
+
if agent not in ALLOWED_AGENTS:
|
|
268
|
+
raise ValueError(f"invalid agent: {agent!r}")
|
|
269
|
+
if agent_role not in ALLOWED_AGENT_ROLES:
|
|
270
|
+
raise ValueError(f"invalid agentRole: {agent_role!r}")
|
|
271
|
+
p = Path(sidecar_path)
|
|
272
|
+
if not p.exists():
|
|
273
|
+
return 0
|
|
274
|
+
payload = json.loads(p.read_text())
|
|
275
|
+
schema = payload.get("schemaVersion")
|
|
276
|
+
if schema not in SUPPORTED_SIDECAR_SCHEMA_VERSIONS:
|
|
277
|
+
raise ValueError(f"unsupported sidecar schemaVersion: {schema!r}")
|
|
278
|
+
entries = payload.get("errors") or []
|
|
279
|
+
recorded_at = _iso(now or _now_utc())
|
|
280
|
+
count = 0
|
|
281
|
+
for e in entries:
|
|
282
|
+
et = e.get("errorType")
|
|
283
|
+
if et not in ALLOWED_ERROR_TYPES:
|
|
284
|
+
raise ValueError(f"invalid errorType in sidecar: {et!r}")
|
|
285
|
+
entry_context = normalize_cause_context(
|
|
286
|
+
e.get("context"), message=e.get("message")
|
|
287
|
+
)
|
|
288
|
+
rec = {
|
|
289
|
+
"ts": e.get("ts"),
|
|
290
|
+
"recordedAt": recorded_at,
|
|
291
|
+
"taskKey": task_key,
|
|
292
|
+
"phase": str(e.get("phase")) if e.get("phase") is not None else None,
|
|
293
|
+
"agent": agent,
|
|
294
|
+
"agentRole": agent_role,
|
|
295
|
+
"model": model,
|
|
296
|
+
"source": "worker-reported",
|
|
297
|
+
"errorType": et,
|
|
298
|
+
"command": e.get("command"),
|
|
299
|
+
"commandKind": e.get("commandKind"),
|
|
300
|
+
"exitCode": e.get("exitCode"),
|
|
301
|
+
"durationMs": e.get("durationMs"),
|
|
302
|
+
"message": e.get("message"),
|
|
303
|
+
"stderrExcerpt": truncate_stderr(e.get("stderrExcerpt")),
|
|
304
|
+
"context": entry_context,
|
|
305
|
+
}
|
|
306
|
+
append_jsonl_line(out_path, rec)
|
|
307
|
+
count += 1
|
|
308
|
+
return count
|
|
@@ -12,7 +12,10 @@ import json
|
|
|
12
12
|
import sys
|
|
13
13
|
from pathlib import Path
|
|
14
14
|
|
|
15
|
-
from okstra_ctl.worker_audit_ledger import
|
|
15
|
+
from okstra_ctl.worker_audit_ledger import (
|
|
16
|
+
check_worker_results_audit,
|
|
17
|
+
worker_result_files,
|
|
18
|
+
)
|
|
16
19
|
|
|
17
20
|
|
|
18
21
|
def _parser() -> argparse.ArgumentParser:
|
|
@@ -26,7 +29,8 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
26
29
|
parser.add_argument("--seq", required=True,
|
|
27
30
|
help="this run's 3-digit seq")
|
|
28
31
|
parser.add_argument("--worker", default=None,
|
|
29
|
-
help="check only this worker id
|
|
32
|
+
help="check only this worker id, with or without the "
|
|
33
|
+
"`-worker` suffix (default: every worker)")
|
|
30
34
|
return parser
|
|
31
35
|
|
|
32
36
|
|
|
@@ -35,8 +39,26 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
35
39
|
failures = check_worker_results_audit(
|
|
36
40
|
args.run_dir, args.task_type, args.seq, worker=args.worker
|
|
37
41
|
)
|
|
38
|
-
|
|
39
|
-
|
|
42
|
+
# A selector that narrows to nothing also produces no failures, so `ok`
|
|
43
|
+
# alone cannot tell a real pass from a check that judged zero files —
|
|
44
|
+
# a mistyped `--worker` used to read as a clean bill of health. The count
|
|
45
|
+
# is what makes the two distinguishable at a glance.
|
|
46
|
+
inspected = [
|
|
47
|
+
path.name
|
|
48
|
+
for path, _role, _seq in worker_result_files(
|
|
49
|
+
args.run_dir, args.task_type, args.seq, args.worker
|
|
50
|
+
)
|
|
51
|
+
]
|
|
52
|
+
print(json.dumps(
|
|
53
|
+
{
|
|
54
|
+
"ok": not failures,
|
|
55
|
+
"inspected": len(inspected),
|
|
56
|
+
"inspectedFiles": inspected,
|
|
57
|
+
"failures": failures,
|
|
58
|
+
},
|
|
59
|
+
ensure_ascii=False,
|
|
60
|
+
indent=2,
|
|
61
|
+
))
|
|
40
62
|
return 2 if failures else 0
|
|
41
63
|
|
|
42
64
|
|