okstra 0.164.0 → 0.165.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/docs/architecture.md +12 -8
- package/docs/cli.md +7 -3
- package/docs/for-ai/README.md +2 -2
- package/docs/for-ai/skills/okstra-inspect.md +2 -2
- package/docs/for-ai/skills/okstra-user-response.md +2 -2
- package/docs/project-structure-overview.md +15 -9
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/antigravity-worker.md +9 -7
- package/runtime/agents/workers/codex-worker.md +9 -7
- package/runtime/agents/workers/grok-worker.md +6 -4
- package/runtime/agents/workers/kimi-worker.md +6 -4
- package/runtime/bin/okstra-antigravity-exec.sh +1 -340
- package/runtime/bin/okstra-claude-exec.sh +1 -178
- package/runtime/bin/okstra-codex-exec.sh +1 -467
- package/runtime/bin/okstra-provider-exec.py +165 -190
- package/runtime/bin/okstra-trace-cleanup.sh +14 -7
- package/runtime/bin/okstra-wrapper-status.py +26 -19
- package/runtime/prompts/lead/adapters/cmux.md +1 -1
- package/runtime/prompts/lead/convergence.md +36 -8
- package/runtime/prompts/lead/okstra-lead-contract.md +23 -1
- package/runtime/prompts/lead/plan-body-verification.md +9 -1
- package/runtime/prompts/lead/report-writer.md +1 -0
- package/runtime/prompts/lead/team-contract.md +3 -3
- package/runtime/prompts/profiles/_common-contract.md +9 -1
- package/runtime/prompts/profiles/_coverage-critic.md +1 -1
- package/runtime/prompts/profiles/_implementation-diff-review.md +3 -1
- package/runtime/prompts/profiles/_implementation-self-check.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +3 -1
- package/runtime/prompts/profiles/implementation-planning.md +5 -3
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +148 -0
- package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +55 -0
- package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +41 -0
- package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +44 -0
- package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +42 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +5 -1
- package/runtime/python/okstra_ctl/dispatch_state.py +10 -0
- package/runtime/python/okstra_ctl/domain/provider.py +5 -1
- package/runtime/python/okstra_ctl/domain/worker_exec.py +102 -0
- package/runtime/python/okstra_ctl/domain/worker_role.py +34 -0
- package/runtime/python/okstra_ctl/domain/worker_stream.py +261 -0
- package/runtime/python/okstra_ctl/incremental_scope.py +16 -4
- package/runtime/python/okstra_ctl/report_html/common.py +71 -25
- package/runtime/python/okstra_ctl/report_html/models.py +5 -0
- package/runtime/python/okstra_ctl/report_html/render.py +1 -1
- package/runtime/python/okstra_ctl/report_html/run_usage.py +19 -0
- package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -0
- package/runtime/python/okstra_ctl/report_views.py +44 -16
- package/runtime/python/okstra_ctl/stage_citations.py +52 -15
- package/runtime/python/okstra_ctl/user_response.py +45 -29
- package/runtime/python/okstra_ctl/wizard.py +13 -9
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +10 -3
- package/runtime/python/okstra_ctl/worker_request.py +140 -0
- package/runtime/python/okstra_ctl/worker_runner.py +622 -0
- package/runtime/python/okstra_token_usage/collect.py +8 -1
- package/runtime/python/okstra_token_usage/report.py +42 -0
- package/runtime/python/okstra_token_usage/task_totals.py +88 -0
- package/runtime/schemas/final-report-v1.0.schema.json +70 -0
- package/runtime/schemas/final-report-v2.0.schema.json +90 -0
- package/runtime/skills/okstra-inspect/SKILL.md +1 -2
- package/runtime/skills/okstra-inspect/facets/logs.md +5 -5
- package/runtime/skills/okstra-inspect/facets/run-audit.md +3 -3
- package/runtime/skills/okstra-run/SKILL.md +1 -1
- package/runtime/skills/okstra-user-response/SKILL.md +15 -5
- package/runtime/templates/report-writer-prompt-preamble.md +1 -0
- package/runtime/templates/reports/html/assets/base.css +8 -4
- package/runtime/templates/reports/html/base.template.html +12 -6
- package/runtime/templates/reports/html/i18n/en.json +29 -6
- package/runtime/templates/reports/html/i18n/ko.json +29 -6
- package/runtime/templates/reports/html/macros/forms.html +9 -3
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +14 -19
- package/runtime/templates/reports/report.js +59 -26
- package/runtime/templates/reports/user-response.template.md +12 -8
- package/runtime/validators/validate-run.py +88 -7
- package/runtime/validators/validate_session_conformance.py +62 -1
- package/src/cli-registry.mjs +0 -7
- package/runtime/bin/okstra-wrapper-agy-stream.py +0 -61
- package/runtime/python/okstra_ctl/error_issue.py +0 -640
- package/runtime/python/okstra_ctl/issue_signals.py +0 -186
- package/runtime/skills/okstra-inspect/facets/error-issue.md +0 -77
- package/src/commands/inspect/error-issue.mjs +0 -27
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
"""Bundled Grok provider catalog."""
|
|
2
2
|
from okstra_ctl.domain.provider import LeadLaunchSpec, ModelSpec, ProviderSpec
|
|
3
|
+
from okstra_ctl.domain.worker_exec import (
|
|
4
|
+
STREAM_JSON,
|
|
5
|
+
ExecCommand,
|
|
6
|
+
PolicySupport,
|
|
7
|
+
WorkerExecRequest,
|
|
8
|
+
)
|
|
9
|
+
from okstra_ctl.domain.worker_stream import content_block_events
|
|
3
10
|
|
|
4
11
|
|
|
5
12
|
GROK = {
|
|
@@ -16,6 +23,42 @@ GROK_LEAD_LAUNCH = LeadLaunchSpec(
|
|
|
16
23
|
)
|
|
17
24
|
|
|
18
25
|
|
|
26
|
+
class GrokExecution:
|
|
27
|
+
"""grok CLI invocation.
|
|
28
|
+
|
|
29
|
+
Runs inside the stage tree when there is one. That is where this provider's
|
|
30
|
+
production path already puts it, and it is why the working directory is part
|
|
31
|
+
of the returned command rather than something the runner picks.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def build_command(self, request: WorkerExecRequest) -> ExecCommand:
|
|
35
|
+
cwd = request.worktree_path or request.project_root
|
|
36
|
+
return ExecCommand(
|
|
37
|
+
argv=(
|
|
38
|
+
"grok",
|
|
39
|
+
"-p",
|
|
40
|
+
request.prompt_text,
|
|
41
|
+
"-m",
|
|
42
|
+
request.model,
|
|
43
|
+
"--output-format",
|
|
44
|
+
"streaming-json",
|
|
45
|
+
"--cwd",
|
|
46
|
+
str(cwd),
|
|
47
|
+
),
|
|
48
|
+
stdin_text=None,
|
|
49
|
+
stream_format=STREAM_JSON,
|
|
50
|
+
normalise=content_block_events,
|
|
51
|
+
cwd=cwd,
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
def policy_support(self) -> PolicySupport:
|
|
55
|
+
return PolicySupport(
|
|
56
|
+
can_auto_approve=False,
|
|
57
|
+
can_bound_write_scope=False,
|
|
58
|
+
note="this CLI exposes no approval or sandbox flag",
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
19
62
|
def create_provider() -> ProviderSpec:
|
|
20
63
|
return ProviderSpec(
|
|
21
64
|
provider="grok",
|
|
@@ -29,4 +72,5 @@ def create_provider() -> ProviderSpec:
|
|
|
29
72
|
wrapper="okstra-grok-exec.sh",
|
|
30
73
|
supported_roles=frozenset({"lead", "analyser", "critic"}),
|
|
31
74
|
lead_launch=GROK_LEAD_LAUNCH,
|
|
75
|
+
exec_strategy=GrokExecution(),
|
|
32
76
|
)
|
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
"""Bundled Kimi provider catalog."""
|
|
2
2
|
from okstra_ctl.domain.provider import LeadLaunchSpec, ModelSpec, ProviderSpec
|
|
3
|
+
from okstra_ctl.domain.worker_exec import (
|
|
4
|
+
STREAM_JSON,
|
|
5
|
+
ExecCommand,
|
|
6
|
+
PolicySupport,
|
|
7
|
+
WorkerExecRequest,
|
|
8
|
+
)
|
|
9
|
+
from okstra_ctl.domain.worker_stream import content_block_events
|
|
3
10
|
|
|
4
11
|
|
|
5
12
|
KIMI = {
|
|
@@ -22,6 +29,40 @@ KIMI_LEAD_LAUNCH = LeadLaunchSpec(
|
|
|
22
29
|
)
|
|
23
30
|
|
|
24
31
|
|
|
32
|
+
class KimiExecution:
|
|
33
|
+
"""kimi CLI invocation.
|
|
34
|
+
|
|
35
|
+
No flag names the working directory, so the process's own is the only
|
|
36
|
+
control. Returning it in the command keeps that fact inside the contract
|
|
37
|
+
instead of leaving the runner to infer it.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def build_command(self, request: WorkerExecRequest) -> ExecCommand:
|
|
41
|
+
cwd = request.worktree_path or request.project_root
|
|
42
|
+
return ExecCommand(
|
|
43
|
+
argv=(
|
|
44
|
+
"kimi",
|
|
45
|
+
"-p",
|
|
46
|
+
request.prompt_text,
|
|
47
|
+
"-m",
|
|
48
|
+
request.model,
|
|
49
|
+
"--output-format",
|
|
50
|
+
"stream-json",
|
|
51
|
+
),
|
|
52
|
+
stdin_text=None,
|
|
53
|
+
stream_format=STREAM_JSON,
|
|
54
|
+
normalise=content_block_events,
|
|
55
|
+
cwd=cwd,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
def policy_support(self) -> PolicySupport:
|
|
59
|
+
return PolicySupport(
|
|
60
|
+
can_auto_approve=False,
|
|
61
|
+
can_bound_write_scope=False,
|
|
62
|
+
note="this CLI exposes no approval or sandbox flag",
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
25
66
|
def create_provider() -> ProviderSpec:
|
|
26
67
|
return ProviderSpec(
|
|
27
68
|
provider="kimi",
|
|
@@ -35,4 +76,5 @@ def create_provider() -> ProviderSpec:
|
|
|
35
76
|
wrapper="okstra-kimi-exec.sh",
|
|
36
77
|
supported_roles=frozenset({"lead", "analyser", "critic"}),
|
|
37
78
|
lead_launch=KIMI_LEAD_LAUNCH,
|
|
79
|
+
exec_strategy=KimiExecution(),
|
|
38
80
|
)
|
|
@@ -4,7 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
import json
|
|
5
5
|
import subprocess
|
|
6
6
|
import time
|
|
7
|
-
from dataclasses import dataclass, field
|
|
7
|
+
from dataclasses import dataclass, field, replace
|
|
8
8
|
from datetime import datetime, timezone
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
from typing import Any, Mapping, Sequence
|
|
@@ -620,6 +620,10 @@ def _refuse_when_cmux_is_walled_off() -> None:
|
|
|
620
620
|
|
|
621
621
|
|
|
622
622
|
def _run_cli_wrapper(plan: DispatchPlan, job: WorkerJob, degraded_from: str) -> WorkerHandle:
|
|
623
|
+
# Whatever the job was planned as, this is a cli-wrapper dispatch: no pane
|
|
624
|
+
# took it, so the worker's stdout is read by whoever called the lead. The
|
|
625
|
+
# record already says so (`_dispatch_record`), and the command must agree.
|
|
626
|
+
job = replace(job, backend=BACKEND_CLI_WRAPPER)
|
|
623
627
|
completed = subprocess.run(job.command, cwd=plan.project_root, text=True)
|
|
624
628
|
return WorkerHandle(job, "", completed, status_path_for_prompt(job.prompt_path), degraded_from)
|
|
625
629
|
|
|
@@ -30,6 +30,7 @@ from .worker_prompt_contract import (
|
|
|
30
30
|
validate_initial_prompt_records,
|
|
31
31
|
validate_reverify_prompt,
|
|
32
32
|
)
|
|
33
|
+
from .worker_runner import LIVE, QUIET
|
|
33
34
|
|
|
34
35
|
BACKEND_CLI_WRAPPER = "cli-wrapper"
|
|
35
36
|
BACKEND_TMUX_PANE = "tmux-pane"
|
|
@@ -91,6 +92,7 @@ class WorkerJob:
|
|
|
91
92
|
|
|
92
93
|
@property
|
|
93
94
|
def command(self) -> list[str]:
|
|
95
|
+
# The flag trails the positional contract the entrypoint reads by index.
|
|
94
96
|
return [
|
|
95
97
|
str(self.wrapper_path),
|
|
96
98
|
str(self.project_root),
|
|
@@ -99,8 +101,16 @@ class WorkerJob:
|
|
|
99
101
|
self.worktree_path,
|
|
100
102
|
self.role,
|
|
101
103
|
str(self.idle_timeout_seconds),
|
|
104
|
+
"--presentation",
|
|
105
|
+
self._presentation(),
|
|
102
106
|
]
|
|
103
107
|
|
|
108
|
+
def _presentation(self) -> str:
|
|
109
|
+
# A pane is a screen a person watches; a cli-wrapper dispatch's stdout is
|
|
110
|
+
# a subagent's context window. Progress belongs in the first and not the
|
|
111
|
+
# second.
|
|
112
|
+
return QUIET if self.backend == BACKEND_CLI_WRAPPER else LIVE
|
|
113
|
+
|
|
104
114
|
def to_payload(self) -> dict[str, Any]:
|
|
105
115
|
return {
|
|
106
116
|
"workerId": self.worker_id,
|
|
@@ -2,7 +2,10 @@
|
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
from dataclasses import dataclass
|
|
5
|
-
from typing import Mapping, Optional
|
|
5
|
+
from typing import TYPE_CHECKING, Mapping, Optional
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from okstra_ctl.domain.worker_exec import ExecutionStrategy
|
|
6
9
|
|
|
7
10
|
|
|
8
11
|
@dataclass(frozen=True)
|
|
@@ -35,6 +38,7 @@ class ProviderSpec:
|
|
|
35
38
|
wrapper: str
|
|
36
39
|
supported_roles: frozenset[str]
|
|
37
40
|
lead_launch: Optional[LeadLaunchSpec] = None
|
|
41
|
+
exec_strategy: Optional["ExecutionStrategy"] = None
|
|
38
42
|
|
|
39
43
|
|
|
40
44
|
@dataclass(frozen=True)
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Provider-neutral execution contracts shared by the dispatcher and workers.
|
|
2
|
+
|
|
3
|
+
A worker run is decided along three axes: which terminal surface it lands on,
|
|
4
|
+
which provider CLI runs it, and which role it plays. This module owns the
|
|
5
|
+
provider axis' vocabulary plus the one rule that is identical for every worker
|
|
6
|
+
regardless of provider — see ``ExecutionPolicy``.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Protocol, runtime_checkable
|
|
13
|
+
|
|
14
|
+
from .worker_stream import Normalise, no_events
|
|
15
|
+
|
|
16
|
+
# What ``ExecCommand.stream_format`` may hold. The value decides whether the
|
|
17
|
+
# runner pipes the CLI's output through the stream formatter or forwards it
|
|
18
|
+
# unchanged, so it is part of the contract rather than a display hint.
|
|
19
|
+
STREAM_JSON = "stream-json"
|
|
20
|
+
TEXT = "text"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class ExecutionPolicy:
|
|
25
|
+
"""How a non-interactive worker is allowed to act.
|
|
26
|
+
|
|
27
|
+
Identical for every worker: nobody is at the keyboard to answer an approval
|
|
28
|
+
prompt, so the gate has to be open and the boundary has to come from a
|
|
29
|
+
sandbox instead. Declared once here rather than per provider, because a
|
|
30
|
+
policy that lives in one adapter file per provider drifts into one policy
|
|
31
|
+
per provider.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
auto_approve: bool
|
|
35
|
+
write_scope: tuple[Path, ...]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class WorkerExecRequest:
|
|
40
|
+
"""One worker dispatch, before any provider has looked at it.
|
|
41
|
+
|
|
42
|
+
``worktree_path`` is the stage tree this run was given, or None for a
|
|
43
|
+
dispatch that works in the project itself. It is kept apart from
|
|
44
|
+
``policy.write_scope`` because the two answer different questions: the scope
|
|
45
|
+
is what may be written, while this is where the run belongs. Providers
|
|
46
|
+
disagree on that second answer, so the request carries both and each
|
|
47
|
+
strategy decides.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
prompt_text: str
|
|
51
|
+
model: str
|
|
52
|
+
project_root: Path
|
|
53
|
+
worktree_path: Path | None
|
|
54
|
+
policy: ExecutionPolicy
|
|
55
|
+
idle_timeout_seconds: int
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True)
|
|
59
|
+
class ExecCommand:
|
|
60
|
+
"""One provider invocation, fully resolved.
|
|
61
|
+
|
|
62
|
+
``stdin_text`` is None when the provider takes its prompt as an argument
|
|
63
|
+
rather than on stdin.
|
|
64
|
+
|
|
65
|
+
``cwd`` is part of the contract because it is not derivable from the argv:
|
|
66
|
+
some CLIs name their working directory in a flag, others inherit the
|
|
67
|
+
process's, and the two groups do not choose the same directory. A runner
|
|
68
|
+
that had to re-derive it would be guessing at what the strategy already
|
|
69
|
+
decided.
|
|
70
|
+
|
|
71
|
+
``normalise`` is the companion of ``stream_format``: declaring a JSON stream
|
|
72
|
+
without saying how to read it is what left one provider's pane blank for a
|
|
73
|
+
whole run. The two are named together so a provider whose events are shaped
|
|
74
|
+
differently cannot be silently handed a formatter that cannot see them.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
argv: tuple[str, ...]
|
|
78
|
+
stdin_text: str | None
|
|
79
|
+
stream_format: str
|
|
80
|
+
cwd: Path
|
|
81
|
+
normalise: Normalise = field(default=no_events)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class PolicySupport:
|
|
86
|
+
"""Whether a provider CLI can express ``ExecutionPolicy`` at all.
|
|
87
|
+
|
|
88
|
+
A provider that cannot must say so rather than quietly accepting the policy
|
|
89
|
+
and running without it. ``note`` carries the reason and is what the contract
|
|
90
|
+
test asserts on.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
can_auto_approve: bool
|
|
94
|
+
can_bound_write_scope: bool
|
|
95
|
+
note: str = ""
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@runtime_checkable
|
|
99
|
+
class ExecutionStrategy(Protocol):
|
|
100
|
+
def build_command(self, request: WorkerExecRequest) -> ExecCommand: ...
|
|
101
|
+
|
|
102
|
+
def policy_support(self) -> PolicySupport: ...
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Per-role execution parameters.
|
|
2
|
+
|
|
3
|
+
Roles differ by value, not by algorithm, so this is a value object rather than
|
|
4
|
+
a strategy. Owning the values here takes the final say back from the shell: the
|
|
5
|
+
dispatcher passed a global idle budget while each wrapper quietly overrode it
|
|
6
|
+
with its own `case "$role"` block.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
|
|
12
|
+
# Implementation executor / verifier dispatches run whole build + test suites
|
|
13
|
+
# that legitimately emit nothing for minutes; the 600s cap reaped them mid-suite
|
|
14
|
+
# (observed: a silent jest + tsc build TERM'd at ~929s). This stays below the
|
|
15
|
+
# 1800s wall-clock polling cap so a genuine hang is still reaped first.
|
|
16
|
+
_BUILD_RUNNING_IDLE_SECONDS = 1500
|
|
17
|
+
_DEFAULT_IDLE_SECONDS = 600
|
|
18
|
+
|
|
19
|
+
_BUILD_RUNNING_ROLES = frozenset({"executor", "verifier"})
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class WorkerRoleSpec:
|
|
24
|
+
role: str
|
|
25
|
+
idle_timeout_seconds: int
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def role_spec(role: str) -> WorkerRoleSpec:
|
|
29
|
+
seconds = (
|
|
30
|
+
_BUILD_RUNNING_IDLE_SECONDS
|
|
31
|
+
if role in _BUILD_RUNNING_ROLES
|
|
32
|
+
else _DEFAULT_IDLE_SECONDS
|
|
33
|
+
)
|
|
34
|
+
return WorkerRoleSpec(role=role, idle_timeout_seconds=seconds)
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
"""Turn a worker's progress into lines a person can read.
|
|
2
|
+
|
|
3
|
+
Pure transformation: no files, no sockets, no clock. The runner decides where
|
|
4
|
+
the lines go; this module decides what they say.
|
|
5
|
+
|
|
6
|
+
Providers do not agree on a wire format, and this layer may not name one. So the
|
|
7
|
+
formatter never reads a provider's JSON: it reads the normalised events below,
|
|
8
|
+
and each provider adapter supplies the function that produces them from its own
|
|
9
|
+
wire shape (``ExecCommand.normalise``). One wire shape is shared widely enough
|
|
10
|
+
to be worth a default here — ``content_block_events``, whose events are keyed on
|
|
11
|
+
``type`` and carry ``message.content`` blocks — but it is a schema, not a
|
|
12
|
+
provider, and an adapter that speaks something else says so in its own file.
|
|
13
|
+
|
|
14
|
+
Screen output folds what the log keeps in full — a tool call becomes one line, a
|
|
15
|
+
tool result becomes an outcome and a size. Thinking is dropped at normalisation,
|
|
16
|
+
which is safe only because worker liveness is measured from stream arrival
|
|
17
|
+
rather than from log writes (see the spec's 7.2).
|
|
18
|
+
|
|
19
|
+
Three projections of the same stream: ``format_live`` for the pane,
|
|
20
|
+
``format_log`` for the archive, and ``final_text`` for the caller that must
|
|
21
|
+
receive the closing message without the progress that produced it.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from typing import Any, Callable, Mapping
|
|
27
|
+
|
|
28
|
+
# Worker panes sit in a two-column grid whose narrowest allowed pane is the
|
|
29
|
+
# floor named by the placement module (60 columns at the time of writing).
|
|
30
|
+
# Holding a summary to two such lines keeps one tool call from pushing the
|
|
31
|
+
# previous one off the top of a short pane.
|
|
32
|
+
_MAX_SUMMARY = 120
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class Text:
|
|
37
|
+
"""Prose the worker addressed to whoever is reading."""
|
|
38
|
+
|
|
39
|
+
body: str
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class ToolCall:
|
|
44
|
+
"""A tool the worker invoked, and the argument worth naming on one row."""
|
|
45
|
+
|
|
46
|
+
name: str
|
|
47
|
+
detail: str = ""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class ToolResult:
|
|
52
|
+
"""What a tool returned.
|
|
53
|
+
|
|
54
|
+
``size_bytes`` is carried rather than derived from ``body`` because the two
|
|
55
|
+
answer different questions: the size is how much the tool produced, while
|
|
56
|
+
the body is what the archive is willing to keep of it.
|
|
57
|
+
|
|
58
|
+
``failed`` is None when the stream closed the tool step without reporting an
|
|
59
|
+
outcome at all. Reading that absence as success would put a confident "ok"
|
|
60
|
+
next to a command that failed.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
body: str
|
|
64
|
+
size_bytes: int
|
|
65
|
+
failed: bool | None = None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass(frozen=True)
|
|
69
|
+
class Denial:
|
|
70
|
+
"""A tool call the provider refused before the worker could make it."""
|
|
71
|
+
|
|
72
|
+
tool: str
|
|
73
|
+
reason: str
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True)
|
|
77
|
+
class Result:
|
|
78
|
+
"""The worker's closing message. At most one per run."""
|
|
79
|
+
|
|
80
|
+
text: str
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
StreamEvent = Text | ToolCall | ToolResult | Denial | Result
|
|
84
|
+
|
|
85
|
+
# What an adapter hands the runner alongside its stream format: one line of the
|
|
86
|
+
# CLI's output, already parsed, turned into however many normalised events it
|
|
87
|
+
# carries. Returning nothing is how an event is dropped.
|
|
88
|
+
Normalise = Callable[[Mapping[str, Any]], tuple[StreamEvent, ...]]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def no_events(event: Mapping[str, Any]) -> tuple[StreamEvent, ...]:
|
|
92
|
+
"""The normaliser for a CLI that emits no events at all.
|
|
93
|
+
|
|
94
|
+
A text CLI's output is already what a person reads, so the runner forwards
|
|
95
|
+
it whole and never calls this. It exists so ``ExecCommand`` can default to
|
|
96
|
+
a truthful "there is nothing here to normalise" instead of to some
|
|
97
|
+
provider's schema.
|
|
98
|
+
"""
|
|
99
|
+
return ()
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def content_block_events(event: Mapping[str, Any]) -> tuple[StreamEvent, ...]:
|
|
103
|
+
"""Normalise the stream shape keyed on ``type`` with ``message.content``.
|
|
104
|
+
|
|
105
|
+
Assistant events carry the worker's prose and its tool calls in the same
|
|
106
|
+
content list, so one event becomes several normalised ones. Thinking blocks
|
|
107
|
+
and everything unrecognised are dropped here rather than in the formatter:
|
|
108
|
+
what is worth reading is a property of the stream, not of the projection.
|
|
109
|
+
"""
|
|
110
|
+
kind = event.get("type")
|
|
111
|
+
if kind == "assistant":
|
|
112
|
+
return tuple(
|
|
113
|
+
item
|
|
114
|
+
for block in _content_blocks(event)
|
|
115
|
+
if (item := _assistant_event(block)) is not None
|
|
116
|
+
)
|
|
117
|
+
if kind == "user":
|
|
118
|
+
return tuple(
|
|
119
|
+
_tool_result(block)
|
|
120
|
+
for block in _content_blocks(event)
|
|
121
|
+
if block.get("type") == "tool_result"
|
|
122
|
+
)
|
|
123
|
+
if kind == "system" and event.get("subtype") == "permission_denied":
|
|
124
|
+
return (
|
|
125
|
+
Denial(
|
|
126
|
+
tool=str(event.get("tool_name", "?")),
|
|
127
|
+
reason=str(event.get("decision_reason", "")).strip(),
|
|
128
|
+
),
|
|
129
|
+
)
|
|
130
|
+
if kind == "result" and isinstance(event.get("result"), str):
|
|
131
|
+
return (Result(text=str(event["result"])),)
|
|
132
|
+
return ()
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def format_live(event: StreamEvent) -> list[str]:
|
|
136
|
+
return _rows(event, limit=_MAX_SUMMARY, include_body=False)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def format_log(event: StreamEvent) -> list[str]:
|
|
140
|
+
"""What the archive keeps: everything the screen shows, plus the bodies.
|
|
141
|
+
|
|
142
|
+
The log is no longer a live window — the worker pane is. It is read after
|
|
143
|
+
the fact, by a person reconstructing why a worker failed and by
|
|
144
|
+
``okstra log-report``.
|
|
145
|
+
"""
|
|
146
|
+
return _rows(event, limit=None, include_body=True)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def final_text(event: StreamEvent) -> str | None:
|
|
150
|
+
"""The worker's closing message, or None for any other event."""
|
|
151
|
+
return event.text if isinstance(event, Result) else None
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _rows(event: StreamEvent, *, limit: int | None, include_body: bool) -> list[str]:
|
|
155
|
+
if isinstance(event, Text):
|
|
156
|
+
return _body_rows(event.body)
|
|
157
|
+
if isinstance(event, ToolCall):
|
|
158
|
+
head = f"→ {event.name}: {event.detail}" if event.detail else f"→ {event.name}"
|
|
159
|
+
return [_truncate(head, limit)]
|
|
160
|
+
if isinstance(event, ToolResult):
|
|
161
|
+
rows = [f" ← {_outcome(event.failed)} ({event.size_bytes} bytes)"]
|
|
162
|
+
if include_body:
|
|
163
|
+
rows.extend(_body_rows(event.body))
|
|
164
|
+
return rows
|
|
165
|
+
if isinstance(event, Denial):
|
|
166
|
+
return [_truncate(f"!! PERMISSION DENIED — {event.tool}: {event.reason}", limit)]
|
|
167
|
+
# A Result is printed by the runner at the point the run ends, not woven
|
|
168
|
+
# into the progress it interrupts.
|
|
169
|
+
return []
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _outcome(failed: bool | None) -> str:
|
|
173
|
+
if failed is None:
|
|
174
|
+
return "done"
|
|
175
|
+
return "error" if failed else "ok"
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _body_rows(text: str) -> list[str]:
|
|
179
|
+
"""One screen row per line, with every blank row dropped.
|
|
180
|
+
|
|
181
|
+
Blank rows are dropped wherever they sit, not only at the end: a tool that
|
|
182
|
+
double-spaces its output would otherwise take twice the pane height it
|
|
183
|
+
earns, and a body that ends in newlines would pad the archive.
|
|
184
|
+
"""
|
|
185
|
+
return [part for part in text.splitlines() if part.strip()]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _assistant_event(block: Mapping[str, Any]) -> StreamEvent | None:
|
|
189
|
+
if block.get("type") == "text":
|
|
190
|
+
return Text(body=str(block.get("text", "")))
|
|
191
|
+
if block.get("type") == "tool_use":
|
|
192
|
+
return ToolCall(
|
|
193
|
+
name=str(block.get("name", "tool")), detail=_tool_detail(block.get("input"))
|
|
194
|
+
)
|
|
195
|
+
return None
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _tool_detail(payload: Any) -> str:
|
|
199
|
+
if not isinstance(payload, Mapping):
|
|
200
|
+
return ""
|
|
201
|
+
for key in ("command", "file_path", "path", "pattern", "query"):
|
|
202
|
+
value = payload.get(key)
|
|
203
|
+
if value:
|
|
204
|
+
return str(value)
|
|
205
|
+
return ""
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _tool_result(block: Mapping[str, Any]) -> ToolResult:
|
|
209
|
+
body = block.get("content")
|
|
210
|
+
return ToolResult(
|
|
211
|
+
body=_body_text(body),
|
|
212
|
+
size_bytes=_body_size(body),
|
|
213
|
+
failed=bool(block.get("is_error")),
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _body_text(body: Any) -> str:
|
|
218
|
+
"""The tool's own output as one string, whatever shape it arrived in."""
|
|
219
|
+
if isinstance(body, str):
|
|
220
|
+
return body
|
|
221
|
+
if isinstance(body, list):
|
|
222
|
+
return "\n".join(
|
|
223
|
+
str(block.get("text", "")) for block in body if isinstance(block, Mapping)
|
|
224
|
+
)
|
|
225
|
+
return ""
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _body_size(body: Any) -> int:
|
|
229
|
+
"""How much a tool returned, in bytes.
|
|
230
|
+
|
|
231
|
+
The content arrives either as a plain string or as a list of content
|
|
232
|
+
blocks. Counting only the string form reports a five-figure result as
|
|
233
|
+
zero — a confident wrong number, which is worse on a screen than no
|
|
234
|
+
number. Bytes rather than characters because the figure exists to be
|
|
235
|
+
compared against the log this run leaves on disk.
|
|
236
|
+
"""
|
|
237
|
+
if isinstance(body, str):
|
|
238
|
+
return len(body.encode("utf-8"))
|
|
239
|
+
if isinstance(body, list):
|
|
240
|
+
return sum(
|
|
241
|
+
len(str(block.get("text", "")).encode("utf-8"))
|
|
242
|
+
for block in body
|
|
243
|
+
if isinstance(block, Mapping)
|
|
244
|
+
)
|
|
245
|
+
return 0
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _content_blocks(event: Mapping[str, Any]) -> list[Mapping[str, Any]]:
|
|
249
|
+
message = event.get("message")
|
|
250
|
+
if not isinstance(message, Mapping):
|
|
251
|
+
return []
|
|
252
|
+
content = message.get("content")
|
|
253
|
+
if not isinstance(content, list):
|
|
254
|
+
return []
|
|
255
|
+
return [block for block in content if isinstance(block, Mapping)]
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _truncate(line: str, limit: int | None) -> str:
|
|
259
|
+
if limit is None or len(line) <= limit:
|
|
260
|
+
return line
|
|
261
|
+
return line[: limit - 1] + "…"
|
|
@@ -293,20 +293,32 @@ def decide_scope(
|
|
|
293
293
|
if not prev_base_sha or prev_base_sha != cur_base_sha:
|
|
294
294
|
return IncrementalDecision(
|
|
295
295
|
"full", [], [],
|
|
296
|
-
|
|
296
|
+
"the branch this plan was written against has moved on "
|
|
297
|
+
f"({prev_base_sha or 'unrecorded'} -> {cur_base_sha or 'unrecorded'}), "
|
|
298
|
+
"so none of the prior stages can be reused as they stand",
|
|
297
299
|
)
|
|
298
300
|
if not impacted_stages:
|
|
299
|
-
return IncrementalDecision(
|
|
301
|
+
return IncrementalDecision(
|
|
302
|
+
"full", [], [],
|
|
303
|
+
"no stage of the prior plan could be tied to the answers, so there is "
|
|
304
|
+
"nothing to narrow the rework down to",
|
|
305
|
+
)
|
|
300
306
|
all_stages = {num for num, _ in stages}
|
|
301
307
|
closure = downstream_stage_closure(stages, set(impacted_stages))
|
|
302
308
|
if len(closure) * 2 > len(all_stages):
|
|
303
309
|
return IncrementalDecision(
|
|
304
310
|
"full", [], [],
|
|
305
|
-
f"
|
|
311
|
+
f"the answers reach {len(closure)} of the plan's {len(all_stages)} stages — "
|
|
312
|
+
f"past the {int(cutoff_ratio * 100)}% mark where replanning outright costs "
|
|
313
|
+
"less than tracking what carried over",
|
|
306
314
|
)
|
|
307
315
|
reverify = sorted(closure)
|
|
308
316
|
carry = sorted(all_stages - closure)
|
|
309
|
-
return IncrementalDecision(
|
|
317
|
+
return IncrementalDecision(
|
|
318
|
+
"incremental", reverify, carry,
|
|
319
|
+
f"the answers reach {len(reverify)} of the plan's {len(all_stages)} stages; "
|
|
320
|
+
"the rest is reused as written",
|
|
321
|
+
)
|
|
310
322
|
|
|
311
323
|
|
|
312
324
|
def _preview_result(args) -> dict:
|