okstra 0.158.1 → 0.160.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture/storage-model.md +2 -0
  3. package/docs/architecture.md +1 -1
  4. package/docs/cli.md +8 -3
  5. package/docs/for-ai/README.md +2 -2
  6. package/docs/for-ai/skills/okstra-inspect.md +3 -0
  7. package/docs/for-ai/skills/okstra-run.md +2 -1
  8. package/docs/for-ai/skills/okstra-user-response.md +5 -5
  9. package/docs/project-structure-overview.md +5 -1
  10. package/docs/task-process/implementation.md +28 -0
  11. package/package.json +1 -1
  12. package/runtime/BUILD.json +2 -2
  13. package/runtime/agents/workers/report-writer-worker.md +1 -1
  14. package/runtime/bin/okstra-claude-exec.sh +4 -1
  15. package/runtime/prompts/host-orchestration/README.md +18 -0
  16. package/runtime/prompts/host-orchestration/implementation.md +57 -0
  17. package/runtime/prompts/launch.template.md +10 -1
  18. package/runtime/prompts/lead/adapters/claude-code.md +1 -1
  19. package/runtime/prompts/lead/context-loader.md +5 -2
  20. package/runtime/prompts/lead/convergence.md +3 -1
  21. package/runtime/prompts/lead/plan-body-verification.md +21 -2
  22. package/runtime/prompts/lead/report-writer.md +1 -1
  23. package/runtime/prompts/lead/team-contract.md +2 -1
  24. package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
  25. package/runtime/prompts/profiles/_common-contract.md +3 -1
  26. package/runtime/prompts/profiles/implementation-planning.md +2 -0
  27. package/runtime/prompts/profiles/requirements-discovery.md +1 -1
  28. package/runtime/prompts/wizard/prompts.ko.json +3 -0
  29. package/runtime/python/okstra_ctl/clarification_items.py +9 -0
  30. package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
  31. package/runtime/python/okstra_ctl/convergence.py +168 -11
  32. package/runtime/python/okstra_ctl/dispatch_core.py +4 -2
  33. package/runtime/python/okstra_ctl/error_issue.py +640 -0
  34. package/runtime/python/okstra_ctl/error_report.py +56 -0
  35. package/runtime/python/okstra_ctl/error_zip.py +23 -10
  36. package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
  37. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
  38. package/runtime/python/okstra_ctl/issue_signals.py +186 -0
  39. package/runtime/python/okstra_ctl/paths.py +38 -0
  40. package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
  41. package/runtime/python/okstra_ctl/profile_show.py +134 -0
  42. package/runtime/python/okstra_ctl/recap.py +63 -0
  43. package/runtime/python/okstra_ctl/render_final_report.py +11 -62
  44. package/runtime/python/okstra_ctl/report_html/filters.py +6 -1
  45. package/runtime/python/okstra_ctl/report_html/render.py +9 -8
  46. package/runtime/python/okstra_ctl/report_html/run_usage.py +110 -0
  47. package/runtime/python/okstra_ctl/report_html/view_models/error_analysis.py +69 -16
  48. package/runtime/python/okstra_ctl/report_html/visualizations.py +107 -14
  49. package/runtime/python/okstra_ctl/report_translation.py +4 -0
  50. package/runtime/python/okstra_ctl/report_views.py +7 -3
  51. package/runtime/python/okstra_ctl/run.py +41 -2
  52. package/runtime/python/okstra_ctl/run_audit.py +477 -0
  53. package/runtime/python/okstra_ctl/usage_cells.py +47 -0
  54. package/runtime/python/okstra_ctl/user_response.py +25 -10
  55. package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
  56. package/runtime/python/okstra_ctl/wizard.py +64 -10
  57. package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
  58. package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
  59. package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
  60. package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
  61. package/runtime/schemas/final-report-v1.0.schema.json +14 -0
  62. package/runtime/schemas/final-report-v2.0.schema.json +56 -2
  63. package/runtime/skills/okstra-inspect/SKILL.md +3 -1
  64. package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
  65. package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
  66. package/runtime/skills/okstra-run/SKILL.md +28 -10
  67. package/runtime/skills/okstra-user-response/SKILL.md +18 -18
  68. package/runtime/templates/reports/final-report.template.md +4 -0
  69. package/runtime/templates/reports/html/assets/base.css +14 -1
  70. package/runtime/templates/reports/html/base.template.html +42 -0
  71. package/runtime/templates/reports/html/i18n/en.json +30 -1
  72. package/runtime/templates/reports/html/i18n/ko.json +30 -1
  73. package/runtime/templates/reports/html/macros/forms.html +15 -0
  74. package/runtime/templates/reports/html/macros/visualizations.html +3 -2
  75. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
  76. package/runtime/templates/reports/i18n/en.json +2 -0
  77. package/runtime/validators/validate-run.py +331 -208
  78. package/runtime/validators/validate_session_conformance.py +102 -32
  79. package/src/cli-registry.mjs +34 -0
  80. package/src/commands/execute/incremental-scope.mjs +10 -0
  81. package/src/commands/execute/worker-audit-check.mjs +35 -0
  82. package/src/commands/inspect/error-issue.mjs +27 -0
  83. package/src/commands/inspect/profile-show.mjs +29 -0
  84. package/src/commands/inspect/run-audit.mjs +26 -0
@@ -1,5 +1,6 @@
1
1
  <!DOCTYPE html>
2
2
  {% from "html/macros/forms.html" import clarification_responses %}
3
+ {% from "html/macros/layout.html" import row_key %}
3
4
  <html lang="en" data-task-template="{{ taskType }}">
4
5
  <head>
5
6
  <meta charset="utf-8">
@@ -38,6 +39,47 @@
38
39
  </ol>
39
40
  </section>
40
41
  {% endif %}
42
+ {% if runUsage %}
43
+ <section data-report-section="run-usage" data-reader-kind="audit">
44
+ <h2>{{ t('runUsage.heading') }}</h2>
45
+ <p class="run-usage-lede">{{ t('runUsage.intro') }}</p>
46
+ <table>
47
+ <thead><tr>
48
+ <th scope="col">{{ t('runUsage.agent') }}</th>
49
+ <th scope="col" class="figure">{{ t('runUsage.raw-tokens') }}</th>
50
+ <th scope="col" class="figure">{{ t('runUsage.billable-tokens') }}</th>
51
+ <th scope="col" class="figure">{{ t('runUsage.cost') }}</th>
52
+ <th scope="col" class="figure">{{ t('runUsage.duration') }}</th>
53
+ </tr></thead>
54
+ <tbody>
55
+ {% for row in runUsage.rows %}
56
+ <tr>
57
+ {{ row_key(pairs=[(t('runUsage.agent'), row.agent), (t('runUsage.role'), row.role), (t('runUsage.model'), row.model)], status_name=t('runUsage.status'), status_raw=row.status, status_text=row.status | enum_label('workerStatus')) }}
58
+ <td class="figure">{{ row.rawTokens }}{% if row.cliTokens %}<span class="cli-extra">{{ t('runUsage.cli') }} {{ row.cliTokens }}</span>{% endif %}</td>
59
+ <td class="figure">{{ row.billableTokens }}</td>
60
+ <td class="figure">{{ row.cost }}{% if row.cliCost %}<span class="cli-extra">+ {{ t('runUsage.cli') }} {{ row.cliCost }}</span>{% endif %}</td>
61
+ <td class="figure">{{ row.duration }}</td>
62
+ </tr>
63
+ {% endfor %}
64
+ {% if runUsage.unaccounted %}
65
+ <tr>
66
+ <td class="row-key">{{ t('runUsage.unaccounted') }}</td>
67
+ <td class="figure">{{ runUsage.unaccounted.rawTokens }}</td>
68
+ <td class="figure">{{ runUsage.unaccounted.billableTokens }}</td>
69
+ <td class="figure">{{ runUsage.unaccounted.cost }}</td>
70
+ <td class="figure"></td>
71
+ </tr>
72
+ {% endif %}
73
+ </tbody>
74
+ <tfoot>
75
+ <tr><th scope="row">{{ t('runUsage.row-lead') }}</th><td class="figure">{{ runUsage.lead.rawTokens }}</td><td class="figure">{{ runUsage.lead.billableTokens }}</td><td class="figure">{{ runUsage.lead.cost }}</td><td class="figure"></td></tr>
76
+ <tr><th scope="row">{{ t('runUsage.row-workers') }}</th><td class="figure">{{ runUsage.worker.rawTokens }}</td><td class="figure">{{ runUsage.worker.billableTokens }}</td><td class="figure">{{ runUsage.worker.cost }}</td><td class="figure"></td></tr>
77
+ <tr class="grand-total"><th scope="row">{{ t('runUsage.row-total') }}</th><td class="figure">{{ runUsage.grand.rawTokens }}</td><td class="figure">{{ runUsage.grand.billableTokens }}</td><td class="figure">{{ runUsage.grand.cost }}</td><td class="figure"></td></tr>
78
+ {% if runUsage.cliCost %}<tr><th scope="row">{{ t('runUsage.row-cli') }}</th><td class="figure"></td><td class="figure"></td><td class="figure">{{ runUsage.cliCost }}</td><td class="figure"></td></tr>{% endif %}
79
+ </tfoot>
80
+ </table>
81
+ </section>
82
+ {% endif %}
41
83
  </main>
42
84
  <footer class="human-report-footer">
43
85
  <button type="button" data-action="export-user-response">{{ t('base.export-my-answers') }}</button>
@@ -60,6 +60,13 @@
60
60
  "handoffMode": {
61
61
  "whole-task": "Whole task",
62
62
  "stage-group": "Selected stages"
63
+ },
64
+ "workerStatus": {
65
+ "completed": "Completed",
66
+ "error": "Failed",
67
+ "timeout": "Timed out",
68
+ "not-run": "Not run",
69
+ "synthesis-only": "Synthesis only"
63
70
  }
64
71
  },
65
72
  "enumHint": {
@@ -86,6 +93,24 @@
86
93
  "drop-that-file-into": ". Drop that file into",
87
94
  "and-the-next-run-picks-your-answers-up-on-it": "and the next run picks your answers up on its own."
88
95
  },
96
+ "runUsage": {
97
+ "heading": "What this run cost",
98
+ "intro": "Every agent this run dispatched, with what it spent and how long it was working. Agents run alongside each other, so their durations do not add up to the elapsed time in the header. Raw tokens are the volume processed; billable tokens are that same work restated in input-price units, which is what the cost is computed from.",
99
+ "agent": "Agent",
100
+ "role": "Role",
101
+ "model": "Model",
102
+ "status": "Status",
103
+ "raw-tokens": "Raw tokens",
104
+ "billable-tokens": "Billable tokens",
105
+ "cost": "Cost",
106
+ "duration": "Working time",
107
+ "cli": "CLI",
108
+ "unaccounted": "Not attributed to an agent above",
109
+ "row-lead": "Lead",
110
+ "row-workers": "Workers",
111
+ "row-total": "Total",
112
+ "row-cli": "CLI calls (billed separately)"
113
+ },
89
114
  "macros": {
90
115
  "forms": {
91
116
  "review-this-analysis": "Review this analysis",
@@ -107,7 +132,11 @@
107
132
  "answer-as": "Answer as",
108
133
  "questions-waiting-on-you": "Questions waiting on you",
109
134
  "count-questions-waiting-on-you": "{count} questions waiting on you",
110
- "your-answer-to-id": "Your answer to {id}"
135
+ "your-answer-to-id": "Your answer to {id}",
136
+ "recommended": "Recommended",
137
+ "scope-impact": "Scope",
138
+ "added-work": "Added work",
139
+ "direction-change": "Direction change"
111
140
  },
112
141
  "visualizations": {
113
142
  "component": "Component",
@@ -60,6 +60,13 @@
60
60
  "handoffMode": {
61
61
  "whole-task": "태스크 전체",
62
62
  "stage-group": "선택한 stage"
63
+ },
64
+ "workerStatus": {
65
+ "completed": "완료",
66
+ "error": "실패",
67
+ "timeout": "시간 초과",
68
+ "not-run": "미실행",
69
+ "synthesis-only": "종합만 수행"
63
70
  }
64
71
  },
65
72
  "enumHint": {
@@ -86,6 +93,24 @@
86
93
  "drop-that-file-into": ". 그 파일을 여기에 두면",
87
94
  "and-the-next-run-picks-your-answers-up-on-it": "다음 run 이 알아서 답변을 읽어 갑니다."
88
95
  },
96
+ "runUsage": {
97
+ "heading": "이번 실행에 든 비용",
98
+ "intro": "이번 실행이 투입한 에이전트별로 얼마를 썼고 얼마나 오래 일했는지입니다. 에이전트는 서로 겹쳐서 돌기 때문에 각 작업 시간을 더해도 머리말의 소요 시간과 같지 않습니다. 원시 토큰은 처리한 분량이고, 과금 환산 토큰은 같은 작업을 입력 단가 기준으로 환산한 값으로 금액은 이 값에서 나옵니다.",
99
+ "agent": "에이전트",
100
+ "role": "역할",
101
+ "model": "모델",
102
+ "status": "상태",
103
+ "raw-tokens": "원시 토큰",
104
+ "billable-tokens": "과금 환산 토큰",
105
+ "cost": "금액",
106
+ "duration": "작업 시간",
107
+ "cli": "CLI",
108
+ "unaccounted": "위 행에 귀속되지 않은 사용량",
109
+ "row-lead": "리드",
110
+ "row-workers": "워커 합계",
111
+ "row-total": "합계",
112
+ "row-cli": "CLI 호출 (별도 청구)"
113
+ },
89
114
  "macros": {
90
115
  "forms": {
91
116
  "review-this-analysis": "이 분석 검토하기",
@@ -107,7 +132,11 @@
107
132
  "answer-as": "답변 형식",
108
133
  "questions-waiting-on-you": "답변을 기다리는 질문",
109
134
  "count-questions-waiting-on-you": "답변을 기다리는 질문 {count}건",
110
- "your-answer-to-id": "{id}에 대한 답변"
135
+ "your-answer-to-id": "{id}에 대한 답변",
136
+ "recommended": "권장",
137
+ "scope-impact": "범위",
138
+ "added-work": "추가 작업",
139
+ "direction-change": "방향 전환"
111
140
  },
112
141
  "visualizations": {
113
142
  "component": "구성 요소",
@@ -40,6 +40,21 @@
40
40
  <p class="eyebrow">{{ row.id }} · {{ row.kind }}</p>
41
41
  {{ row.statement | paragraphs }}
42
42
  <dl class="clarification-expected"><dt>{{ t('macros.forms.answer-as') }}</dt><dd>{{ row.expectedForm | inline_code }}</dd></dl>
43
+ {% if row.options %}
44
+ <ol class="clarification-options">
45
+ {% for option in row.options %}
46
+ <li class="clarification-option{% if option.role == 'recommended' %} is-recommended{% endif %}">
47
+ <p class="clarification-option-answer">{{ option.answer }}{% if option.role == 'recommended' %} <span class="badge">{{ t('macros.forms.recommended') }}</span>{% endif %}</p>
48
+ <p class="clarification-option-rationale">{{ option.rationale }}</p>
49
+ <dl class="clarification-option-impact">
50
+ <dt>{{ t('macros.forms.scope-impact') }}</dt><dd>{{ option.scopeImpact | join(', ') }}</dd>
51
+ <dt>{{ t('macros.forms.added-work') }}</dt><dd>{{ option.addedWork }}</dd>
52
+ <dt>{{ t('macros.forms.direction-change') }}</dt><dd>{{ option.directionChange }}</dd>
53
+ </dl>
54
+ </li>
55
+ {% endfor %}
56
+ </ol>
57
+ {% endif %}
43
58
  <label for="response-{{ row.id }}">{{ t('macros.forms.your-answer-to-id') | replace('{id}', row.id) }}</label>
44
59
  <textarea id="response-{{ row.id }}" data-response-id="{{ row.id }}" rows="4">{{ row.userInput | default('') }}</textarea>
45
60
  </article>
@@ -7,11 +7,12 @@
7
7
  </figcaption>
8
8
  <div class="visualization" aria-hidden="true">{{ model.svg | safe }}</div>
9
9
  {% set show_paths = model.nodes | selectattr("paths") | first is defined %}
10
+ {% set show_detail = model.nodes | selectattr("detail") | first is defined %}
10
11
  <table class="visualization-fallback">
11
- <thead><tr><th>{{ t('macros.visualizations.component') }}</th><th>{{ t('macros.visualizations.what-it-does') }}</th>{% if show_paths %}<th>{{ t('macros.visualizations.paths') }}</th>{% endif %}</tr></thead>
12
+ <thead><tr><th>{{ t('macros.visualizations.component') }}</th>{% if show_detail %}<th>{{ t('macros.visualizations.what-it-does') }}</th>{% endif %}{% if show_paths %}<th>{{ t('macros.visualizations.paths') }}</th>{% endif %}</tr></thead>
12
13
  <tbody>
13
14
  {% for node in model.nodes %}
14
- <tr{% if anchor_nodes %} id="id-{{ node.id }}"{% endif %} data-fallback-id="{{ node.id }}">{{ row_key(pairs=[("ID", node.id), ("Name", node.label), ("Kind", node.note)]) }}<td>{{ node.detail | inline_code }}</td>{% if show_paths %}<td>{% for path in node.paths %}<code>{{ path }}</code>{% if not loop.last %} {% endif %}{% endfor %}</td>{% endif %}</tr>
15
+ <tr{% if anchor_nodes %} id="id-{{ node.id }}"{% endif %} data-fallback-id="{{ node.id }}">{{ row_key(pairs=[("ID", node.id), ("Name", node.label), ("Kind", node.note)]) }}{% if show_detail %}<td>{{ node.detail | inline_code }}</td>{% endif %}{% if show_paths %}<td>{% for path in node.paths %}<code>{{ path }}</code>{% if not loop.last %} {% endif %}{% endfor %}</td>{% endif %}</tr>
15
16
  {% endfor %}
16
17
  </tbody>
17
18
  </table>
@@ -69,6 +69,7 @@
69
69
  <h2>{{ t('tasks.implementation-planning.plan-body-verification') }}</h2>
70
70
  <p><strong>{{ t('tasks.implementation-planning.verdict') }}</strong> — {{ planning.planBodyVerification.gateResult | inline_code }} ({{ planning.planBodyVerification.roundCount }} rounds{% if planning.planBodyVerification.get("selfFixRoundsApplied") is not none %}, {{ planning.planBodyVerification.selfFixRoundsApplied }} self-fix rounds{% if planning.planBodyVerification.get("selfFixStopReason") %} · {{ planning.planBodyVerification.selfFixStopReason | inline_code }}{% endif %}{% endif %})</p>
71
71
  {% if planning.planBodyVerification.get("gateBlockedBy") %}<p><strong>{{ t('tasks.implementation-planning.what-it-blocked') }}</strong> — {{ planning.planBodyVerification.gateBlockedBy | join(", ") | inline_code }}</p>{% endif %}
72
+ {% if planning.planBodyVerification.get("uniformVerifiers") %}<p><strong>{{ t('implementationPlanning.planBodyUniformVerifierLabel') }}</strong> — {% for u in planning.planBodyVerification.uniformVerifiers %}{{ u.worker | inline_code }} → {{ u.verdict | inline_code }} ({{ u.itemCount }}){% if not loop.last %}, {% endif %}{% endfor %}<br><small>{{ t('implementationPlanning.planBodyUniformVerifierLegend') }}</small></p>{% endif %}
72
73
  <table><thead><tr><th>{{ t('tasks.implementation-planning.plan-item') }}</th><th>{{ t('tasks.implementation-planning.subject') }}</th></tr></thead><tbody>{% for row in planning.planBodyVerification.planItems %}<tr id="id-{{ row.id }}">{{ row_key(pairs=[("ID", row.id), ("Source section", row.sourceSection)]) }}<td>{{ row.subject | inline_code }}</td></tr>{% endfor %}</tbody></table>
73
74
  {% if planning.planBodyVerification.get("dissentLog") %}<h3>{{ t('tasks.implementation-planning.dissent-left-standing') }}</h3>
74
75
  <table><thead><tr><th>{{ t('tasks.implementation-planning.subject') }}</th><th>{{ t('tasks.implementation-planning.body') }}</th></tr></thead><tbody>{% for row in planning.planBodyVerification.dissentLog %}<tr>{{ row_key(pairs=[("Subject", row.planItem), ("Worker", row.workerRole)]) }}<td>{{ row.body | inline_code }}</td></tr>{% endfor %}</tbody></table>{% endif %}
@@ -137,6 +137,8 @@
137
137
  "implementationPlanning": {
138
138
  "planBodyGateLegend": "Gate values — `passed`: agreed, no dissent · `passed-with-dissent`: a minority dissent remains but the gate passes (a majority dissent would block approval) · `blocked-by-disagreement`: majority dissent blocks approval · `aborted-non-result`: verification itself produced no result.",
139
139
  "planBodyBlockedByLegend": "Which input blocked the gate — `majority-disagree`: a worker majority dissent · `coverage-gap`: a Requirement Coverage gap / blocked row, independent of worker votes · `non-result`: verification produced no result. Absent means nothing blocked.",
140
+ "planBodyUniformVerifierLabel": "Uniform verifier",
141
+ "planBodyUniformVerifierLegend": "This verifier answered the same verdict to every item it voted on, so this round's refutation signal came from its peers alone — the gate above reads as a wider cross-check than it was. A unanimous round is a legitimate outcome; confirm from the worker's `-audit-` sidecar that it actually opened the cited evidence.",
140
142
  "planBodyVerdictLegend": "Verdict — **AGREE**: executable as written and internally consistent with other items · **SUPPLEMENT**: item is sound but a dependency / edge case / precondition is missing · **DISAGREE**: has a defect (see Breakage kind) · **verification-error**: the worker produced no result.",
141
143
  "planBodyBreakageLegend": "Breakage kind — a: cited file path/symbol mismatches another step or option · b: command is not executable or is ambiguous · c: validation signal is not observable · d: rollback violates commit/dependency order (advisory — a rollback is human-run, so this never blocks the gate) · e: contradicts the trade-off matrix · f: requirement-coverage row does not map to an option/stage/step that actually satisfies the requirement. (`--` = not applicable) Fixability: planner-fixable = correctable from code + plan + brief; needs-user-input = requires an external decision.",
142
144
  "planBodySourceLabel": "source §",