value-stream 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- value_stream/__init__.py +1 -0
- value_stream/app/SPEC.md +79 -0
- value_stream/app/__init__.py +1 -0
- value_stream/app/__main__.py +28 -0
- value_stream/app/contracts.py +18 -0
- value_stream/app/coordinator.py +359 -0
- value_stream/app/errors.py +17 -0
- value_stream/app/exports.py +142 -0
- value_stream/app/gateway.py +130 -0
- value_stream/app/generation.py +269 -0
- value_stream/app/insights.py +75 -0
- value_stream/app/metrics.py +132 -0
- value_stream/app/openapi.json +3699 -0
- value_stream/app/schemas.py +288 -0
- value_stream/app/server.py +298 -0
- value_stream/app/settings.py +46 -0
- value_stream/app/storage.py +651 -0
- value_stream/client/__init__.py +3 -0
- value_stream/client/simulation_runner.py +83 -0
- value_stream/client/views/__init__.py +4 -0
- value_stream/client/views/metadata_viewer.py +235 -0
- value_stream/client/views/result_viewer.py +76 -0
- value_stream/client/views/viewer.py +8 -0
- value_stream/client/web/__init__.py +10 -0
- value_stream/client/web/client.py +115 -0
- value_stream/client/web/errors.py +29 -0
- value_stream/client/web/local_service.py +92 -0
- value_stream/factory/__init__.py +5 -0
- value_stream/factory/developer_factory.py +33 -0
- value_stream/factory/factory.py +43 -0
- value_stream/factory/task_factory.py +64 -0
- value_stream/factory/task_generator.py +102 -0
- value_stream/policy/__init__.py +3 -0
- value_stream/policy/simulation_policy.py +20 -0
- value_stream/resources/__init__.py +23 -0
- value_stream/resources/developer.py +53 -0
- value_stream/resources/qa_tester.py +51 -0
- value_stream/resources/resource.py +162 -0
- value_stream/resources/resource_metadata.py +24 -0
- value_stream/resources/resource_policy.py +13 -0
- value_stream/resources/resource_pool.py +39 -0
- value_stream/resources/resource_tracker.py +160 -0
- value_stream/resources/toolchain.py +43 -0
- value_stream/service/SPEC.md +46 -0
- value_stream/service/__init__.py +1 -0
- value_stream/service/app.py +218 -0
- value_stream/service/codec.py +237 -0
- value_stream/service/job_store.py +322 -0
- value_stream/service/openapi.json +1283 -0
- value_stream/service/scheduler.py +234 -0
- value_stream/service/schemas.py +248 -0
- value_stream/service/settings.py +37 -0
- value_stream/service/worker.py +82 -0
- value_stream/simulation/__init__.py +16 -0
- value_stream/simulation/default_simulation_policy.py +68 -0
- value_stream/simulation/model.py +30 -0
- value_stream/simulation/model_factory.py +54 -0
- value_stream/simulation/simulation.py +144 -0
- value_stream/simulation/simulation_metadata.py +12 -0
- value_stream/simulation/simulation_result.py +20 -0
- value_stream/task/__init__.py +23 -0
- value_stream/task/epoch.py +21 -0
- value_stream/task/event_status.py +8 -0
- value_stream/task/task.py +260 -0
- value_stream/task/task_event.py +97 -0
- value_stream/task/task_history.py +208 -0
- value_stream/task/task_router.py +70 -0
- value_stream/task/task_state.py +7 -0
- value_stream/task/task_type.py +8 -0
- value_stream/workflow/__init__.py +18 -0
- value_stream/workflow/assignment_strategy.py +10 -0
- value_stream/workflow/pool_manager.py +136 -0
- value_stream/workflow/resource_operator.py +204 -0
- value_stream/workflow/sdlc_workflow.py +99 -0
- value_stream/workflow/support_workflow.py +23 -0
- value_stream/workflow/task_store.py +125 -0
- value_stream/workflow/workflow_policy.py +12 -0
- value_stream-0.1.1.dist-info/METADATA +288 -0
- value_stream-0.1.1.dist-info/RECORD +81 -0
- value_stream-0.1.1.dist-info/WHEEL +4 -0
- value_stream-0.1.1.dist-info/licenses/LICENSE +21 -0
value_stream/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__all__ = []
|
value_stream/app/SPEC.md
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Simulation web application specification
|
|
2
|
+
|
|
3
|
+
The implementation design and validation milestones are to be documented in `.agent/SIMULATION_APP_EXECPLAN.md`, maintained according to `.agent/PLANS.md`. This specification records the agreed behavior; the ExecPlan resolves version 1 implementation choices.
|
|
4
|
+
|
|
5
|
+
## Goals
|
|
6
|
+
- Create a standalone web application that provides a user interface to the API offered by `value_stream.service`.
|
|
7
|
+
- Provide insight on how model properties affect the loss of value for a delivered set of tasks.
|
|
8
|
+
- Identify the source(s) of loss within a simulation.
|
|
9
|
+
- Provide suggestions for reducing loss.
|
|
10
|
+
|
|
11
|
+
## Version 1 requirements
|
|
12
|
+
|
|
13
|
+
- Application source files belong under `src/value_stream/app`.
|
|
14
|
+
|
|
15
|
+
### Collect User Input
|
|
16
|
+
|
|
17
|
+
Create a set of tasks, given: task count, story points, initial value, and depreciation rate. Story points and initial value may be specified as constants or bounded uniform ranges. Initial value defaults to 1, and depreciation rate defaults to 0.005 (0.5% per simulation time unit). Use explicit time-unit labels. There is no separate complexity property; task effort is represented by `Task.story_points`.
|
|
18
|
+
|
|
19
|
+
Create one or more models. Models may differ:
|
|
20
|
+
- In the size and properties of developer teams, QA Tester pools, toolchain pools
|
|
21
|
+
- In the properties of deployment cadence, support interval, support task story points
|
|
22
|
+
- Developers within a team may differ in efficiency. Collect team size, minimum efficiency, maximum efficiency, and a linear or bounded normal distribution. Linear means evenly spaced values between the bounds; a one-person linear team uses the midpoint. Developer teams may differ between models.
|
|
23
|
+
- For V1, QA testers are specified as a pooled resource. Pool size and properties may differ between models.
|
|
24
|
+
- For V1, the toolchain will be specified as a pooled resource with user-specified deployment duration and failure rate.
|
|
25
|
+
|
|
26
|
+
A job is a request to execute a simulation for every model, applying each to the same task set. A task set may be reused across jobs and must remain consistent within a job.
|
|
27
|
+
|
|
28
|
+
Support both individually configured models and parameter sweeps across multiple properties. A sweep generates all combinations of selected values. Sweeps are the primary model-creation workflow; show the model count and validate limits before execution. Support duplicating and editing model definitions without repetitive entry.
|
|
29
|
+
|
|
30
|
+
Give every scenario a stable identity and a human-readable name. Freeze generated task sets and developer teams, and retain generation and execution seeds so reordering, repeating, or selectively rerunning scenarios does not change their numerical outcomes. Provide an explicit task-regeneration action.
|
|
31
|
+
|
|
32
|
+
Utilize best practices in UI design to guide the user and avoid repetitive data entry.
|
|
33
|
+
|
|
34
|
+
### Execute Simulation
|
|
35
|
+
- A set of tasks and one or more models are submitted to the web service for execution as a job.
|
|
36
|
+
- Provide progress updates and update plots as individual models finish. Streaming intermediate results from inside a running model is not required.
|
|
37
|
+
- Cancellation stops the whole job and preserves already completed results. Individual model failures do not prevent other models from completing.
|
|
38
|
+
|
|
39
|
+
### Plot Results
|
|
40
|
+
|
|
41
|
+
- Render results using plots similar to those implemented in `src/value_stream/client/views/metadata_viewer.py` and `result_viewer.py`, including:
|
|
42
|
+
- Loss vs. Cadence
|
|
43
|
+
- Loss vs. Team Size
|
|
44
|
+
- Mean Stage Loss
|
|
45
|
+
- Resource Utilization
|
|
46
|
+
- Resource Backlog (the existing Resource Capacity view)
|
|
47
|
+
|
|
48
|
+
- Plots are updated in realtime as results are returned by the simulation while it is executed.
|
|
49
|
+
- Support switching between plots via labeled, keyboard-accessible tabs.
|
|
50
|
+
- Support pan, zoom, and reset zoom, with consistent scenario colors and accessible alternatives to plot-only information.
|
|
51
|
+
- Display value lost as a positive percentage, with lower values better. Overall value lost is `100 * (initial value - delivered value) / initial value`; show N/A when total initial value is zero. Explain the separate meaning of mean stage loss instead of presenting it as an additive breakdown of overall loss. Resource metric labels must accurately describe the telemetry available.
|
|
52
|
+
- For V1, preserve the resource metrics used by the current viewers. Explain that resource utilization shows recorded activity shares, not a precise measure of total capacity utilization, and that backlog counts resource requests, which can contain task batches. Expanded simulation telemetry is deferred.
|
|
53
|
+
|
|
54
|
+
- Support an interactive mode after a job is executed. Retain the baseline and add the latest interactive comparison, with a Pin comparison action to retain additional comparisons. For example, a team-size sweep produces one series at a fixed cadence; changing cadence produces another series using the same task set and the same teams for each size. This provides an alternative to sweeping cadence in advance. Commit slider changes on release, replace unfinished interactive work when superseded, reuse unchanged results, and keep prior plots visible while updating. Baseline and pinned comparisons remain immutable.
|
|
55
|
+
|
|
56
|
+
- Errors are handled gracefully. When possible, continue if there is an error while executing the simulation, presenting the user with relevant details.
|
|
57
|
+
|
|
58
|
+
- Support separate CSV exports for model summaries, stage events, and resource history, with scenario identities and relevant configuration/seed provenance.
|
|
59
|
+
- Provide explainable, rule-based observations and suggestions for reducing loss, with an explicit action to test a suggested change. Distinguish observed evidence from an untested hypothesis. Automatic optimization and machine learning are deferred.
|
|
60
|
+
|
|
61
|
+
### Other Requirements
|
|
62
|
+
|
|
63
|
+
- Put storage behind a replaceable interface, using an in-memory implementation for V1. Application life means the life of the server: browser refreshes retain task sets, model definitions, and retained comparisons. Document configurable retention limits; state need not survive server shutdown.
|
|
64
|
+
- Include unit and integration tests
|
|
65
|
+
- Handle invalid input, execution failures, and connection failures with a documented error contract.
|
|
66
|
+
- Provide an option to package and run the application as a Docker image, and support local standalone startup from the command line.
|
|
67
|
+
- Add configurable protections against excessive resource use, including admission and execution limits. Select initial defaults during design and document their errors in the API contract.
|
|
68
|
+
- Provide single startup of web application and simulation service, with UI and application API on one browser origin. Also permit an explicitly configured simulation service endpoint.
|
|
69
|
+
- User interface design must be clean, modern, and self-describing.
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
## Intended use and future scope
|
|
74
|
+
|
|
75
|
+
Version 1 is for individuals demonstrating the value-stream project and runs as a single server instance. Horizontal scaling, shared multi-user deployment, authentication, policy extensibility, service discovery, persistent results, and machine-learning optimization are later phases. Keep storage, job coordination, and simulation execution separable, with stable identifiers, to allow future shared storage and distributed workers. V1 does not implement distributed infrastructure or claim multi-instance support.
|
|
76
|
+
|
|
77
|
+
## Implementation design prerequisites
|
|
78
|
+
|
|
79
|
+
- Write an ExecPlan for this significant feature before coding, following `.agent/PLANS.md` as required by `AGENTS.md`.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Interactive, temporary workspaces for value-stream simulation."""
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Run the packaged browser application and its local simulation service."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import uvicorn
|
|
6
|
+
from .server import create_app
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main():
|
|
10
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
11
|
+
parser.add_argument("--host", default="127.0.0.1")
|
|
12
|
+
parser.add_argument("--port", type=int, default=8081)
|
|
13
|
+
parser.add_argument("--service-url", default=None)
|
|
14
|
+
args = parser.parse_args()
|
|
15
|
+
if not (Path(__file__).parent / "static" / "index.html").is_file():
|
|
16
|
+
parser.error(
|
|
17
|
+
"Build the UI first: npm ci --prefix src/value_stream/app/frontend && npm run build --prefix src/value_stream/app/frontend"
|
|
18
|
+
)
|
|
19
|
+
uvicorn.run(
|
|
20
|
+
create_app(service_url=args.service_url),
|
|
21
|
+
host=args.host,
|
|
22
|
+
port=args.port,
|
|
23
|
+
workers=1,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
if __name__ == "__main__":
|
|
28
|
+
main()
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Regenerate both checked-in OpenAPI contracts without starting servers."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from .server import create_app
|
|
6
|
+
from value_stream.service.app import create_app as create_service
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main():
|
|
10
|
+
root = Path(__file__).resolve().parents[1]
|
|
11
|
+
for name, factory in [("app", create_app), ("service", create_service)]:
|
|
12
|
+
(root / name / "openapi.json").write_text(
|
|
13
|
+
json.dumps(factory().openapi(), indent=2, sort_keys=True) + "\n"
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
if __name__ == "__main__":
|
|
18
|
+
main()
|
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
"""Application run coordination, separate from routes and worker execution."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import logging
|
|
7
|
+
import time
|
|
8
|
+
from uuid import uuid5, UUID
|
|
9
|
+
from value_stream.app.schemas import RunStatus
|
|
10
|
+
from value_stream.service.schemas import JobRequest, ErrorEnvelope, ResultData
|
|
11
|
+
from .errors import AppError
|
|
12
|
+
from .generation import concrete_model, validate_model_limits
|
|
13
|
+
from .metrics import loss_percent
|
|
14
|
+
from .schemas import TERMINAL, ModelSettings, RunRequest, Status
|
|
15
|
+
from .storage import WorkspaceStore, RunRecord
|
|
16
|
+
from .gateway import SimulationGateway
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RunCoordinator:
|
|
22
|
+
"""Coordinates workspace runs and communicates with the simulation service."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, store: WorkspaceStore, gateway: SimulationGateway) -> None:
|
|
25
|
+
"""Configure run coordination with its workspace store and simulation gateway.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
store (WorkspaceStore): Store used to retrieve or update data.
|
|
29
|
+
gateway (SimulationGateway): Simulation service gateway.
|
|
30
|
+
"""
|
|
31
|
+
self.store, self.gateway = store, gateway
|
|
32
|
+
self.tasks = {}
|
|
33
|
+
self.closed = False
|
|
34
|
+
|
|
35
|
+
async def start_run(self, workspace_id: UUID, request: RunRequest) -> RunStatus:
|
|
36
|
+
"""Start a run for the requested workspace scenarios.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
workspace_id (UUID): Identifier of the workspace.
|
|
40
|
+
request (RunRequest): Request data to process.
|
|
41
|
+
|
|
42
|
+
Raises:
|
|
43
|
+
AppError: If the service is unavailable, the request is invalid, or an active run conflicts with it.
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
RunStatus: The resulting value.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
if self.closed:
|
|
50
|
+
raise AppError("SERVICE_UNAVAILABLE", "Application is shutting down", 503)
|
|
51
|
+
w = self.store.get(workspace_id)
|
|
52
|
+
previous = self.store.previous(w, request)
|
|
53
|
+
if previous:
|
|
54
|
+
return previous.status
|
|
55
|
+
if request.intent == "interactive":
|
|
56
|
+
if (
|
|
57
|
+
request.source_run_id is None
|
|
58
|
+
or request.property is None
|
|
59
|
+
or request.property not in ModelSettings.model_fields.keys()
|
|
60
|
+
):
|
|
61
|
+
raise AppError("INVALID_INPUT", "Choose a source comparison and a model property")
|
|
62
|
+
source = self.store.run(workspace_id, request.source_run_id)
|
|
63
|
+
if source.status.state not in TERMINAL or not source.results:
|
|
64
|
+
raise AppError("INVALID_INPUT", "Interactive mode requires a completed comparison")
|
|
65
|
+
if w.current_task_set != source.task_set.id:
|
|
66
|
+
raise AppError(
|
|
67
|
+
"REVISION_CONFLICT",
|
|
68
|
+
"Interactive comparisons must use the current task set",
|
|
69
|
+
409,
|
|
70
|
+
)
|
|
71
|
+
scenarios = [
|
|
72
|
+
o.scenario.model_copy(deep=True)
|
|
73
|
+
for o in source.status.outcomes
|
|
74
|
+
if (
|
|
75
|
+
request.definition_id is None
|
|
76
|
+
or o.scenario.definition_id == request.definition_id
|
|
77
|
+
)
|
|
78
|
+
and getattr(o.scenario.settings, request.property) == request.family_value
|
|
79
|
+
]
|
|
80
|
+
if not scenarios:
|
|
81
|
+
raise AppError(
|
|
82
|
+
"INVALID_INPUT",
|
|
83
|
+
"Select an existing value/family before changing it",
|
|
84
|
+
)
|
|
85
|
+
for scenario in scenarios:
|
|
86
|
+
scenario.settings = ModelSettings.model_validate(
|
|
87
|
+
{**scenario.settings.model_dump(), request.property: request.value}
|
|
88
|
+
)
|
|
89
|
+
validate_model_limits(scenario.settings, self.store.settings)
|
|
90
|
+
scenario.model = concrete_model(scenario.settings, scenario.team_seed)
|
|
91
|
+
scenario.revision += 1
|
|
92
|
+
scenario.name = (
|
|
93
|
+
f"{scenario.name.split(' → ')[0]} → {request.property}={request.value}"
|
|
94
|
+
)
|
|
95
|
+
task_set = source.task_set
|
|
96
|
+
# Supersession is explicit and confirmed before replacement admission.
|
|
97
|
+
active = next((r for r in w.runs if r.state not in TERMINAL), None)
|
|
98
|
+
if active:
|
|
99
|
+
if active.intent != "interactive":
|
|
100
|
+
raise AppError("RUN_ACTIVE", "Finish or cancel the manual run first", 409)
|
|
101
|
+
active.superseded = True
|
|
102
|
+
await self.cancel_run(workspace_id, active.id)
|
|
103
|
+
raise AppError(
|
|
104
|
+
"RUN_ACTIVE",
|
|
105
|
+
"Waiting for the previous interactive job to stop",
|
|
106
|
+
409,
|
|
107
|
+
)
|
|
108
|
+
else:
|
|
109
|
+
preview = self.store.preview(workspace_id)
|
|
110
|
+
if preview.digest != request.preview_digest:
|
|
111
|
+
raise AppError(
|
|
112
|
+
"REVISION_CONFLICT",
|
|
113
|
+
"Preview is stale. Preview the current inputs again.",
|
|
114
|
+
409,
|
|
115
|
+
)
|
|
116
|
+
scenarios = preview.scenarios
|
|
117
|
+
task_set = next(t for t in w.task_sets if t.id == preview.task_set_id)
|
|
118
|
+
record, created = self.store.register(workspace_id, request, scenarios, task_set)
|
|
119
|
+
if created:
|
|
120
|
+
self.launch(workspace_id, record)
|
|
121
|
+
return record.status
|
|
122
|
+
|
|
123
|
+
def launch(self, workspace_id: UUID, record: RunRecord) -> None:
|
|
124
|
+
"""Launch execution of a simulation run.
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
workspace_id (UUID): Identifier of the workspace.
|
|
128
|
+
record (RunRecord): Run record to update.
|
|
129
|
+
"""
|
|
130
|
+
record.status.retry_paused = False
|
|
131
|
+
task = asyncio.create_task(self.execute(workspace_id, record))
|
|
132
|
+
self.tasks[record.status.id] = task
|
|
133
|
+
task.add_done_callback(
|
|
134
|
+
lambda done: (
|
|
135
|
+
self.tasks.pop(record.status.id, None)
|
|
136
|
+
if self.tasks.get(record.status.id) is done
|
|
137
|
+
else None
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
def complete_outcome(
|
|
142
|
+
self,
|
|
143
|
+
record: RunRecord,
|
|
144
|
+
index: int,
|
|
145
|
+
status: Status,
|
|
146
|
+
result: Optional[ResultData] = None,
|
|
147
|
+
error: ErrorEnvelope | None = None,
|
|
148
|
+
cached: bool = False,
|
|
149
|
+
) -> None:
|
|
150
|
+
"""Record the completed scenario outcome.
|
|
151
|
+
|
|
152
|
+
Args:
|
|
153
|
+
record (RunRecord): Run record to update.
|
|
154
|
+
index (int): Index of the scenario or model.
|
|
155
|
+
status (Status): Current outcome status.
|
|
156
|
+
result (Optional[ResultData]): Simulation result to retain.
|
|
157
|
+
error (ErrorEnvelope | None): Optional error returned for the scenario.
|
|
158
|
+
cached (bool): Whether the result came from the cache.
|
|
159
|
+
"""
|
|
160
|
+
outcome = record.status.outcomes[index]
|
|
161
|
+
if outcome.cursor:
|
|
162
|
+
return
|
|
163
|
+
if result is not None:
|
|
164
|
+
self.store.retain_result(record, index, result)
|
|
165
|
+
outcome.loss_percent = loss_percent(
|
|
166
|
+
sum(t.initial_value for t in record.task_set.tasks),
|
|
167
|
+
result.summary_result.total_delivered_value,
|
|
168
|
+
)
|
|
169
|
+
outcome.completion_time = result.summary_result.completion_time
|
|
170
|
+
outcome.delivered_value = result.summary_result.total_delivered_value
|
|
171
|
+
outcome.status, outcome.error, outcome.cached = status, error, cached
|
|
172
|
+
record.status.last_cursor += 1
|
|
173
|
+
outcome.cursor = record.status.last_cursor
|
|
174
|
+
|
|
175
|
+
async def execute(self, workspace_id: UUID, record: RunRecord) -> None:
|
|
176
|
+
"""Execute the requested simulation operation.
|
|
177
|
+
|
|
178
|
+
Args:
|
|
179
|
+
workspace_id (UUID): Identifier of the workspace.
|
|
180
|
+
record (RunRecord): Run record to update.
|
|
181
|
+
|
|
182
|
+
Raises:
|
|
183
|
+
AppError: If the simulation service rejects the job or cannot be reached.
|
|
184
|
+
"""
|
|
185
|
+
status = record.status
|
|
186
|
+
retry_start = None
|
|
187
|
+
delay = self.store.settings.poll_seconds
|
|
188
|
+
try:
|
|
189
|
+
for i, key in enumerate(record.cache_keys):
|
|
190
|
+
if i not in record.service_indices and not status.outcomes[i].cursor:
|
|
191
|
+
self.complete_outcome(
|
|
192
|
+
record,
|
|
193
|
+
i,
|
|
194
|
+
Status.SUCCEEDED,
|
|
195
|
+
self.store.cached_result(key),
|
|
196
|
+
cached=True,
|
|
197
|
+
)
|
|
198
|
+
if not record.service_indices:
|
|
199
|
+
status.state = "completed"
|
|
200
|
+
self.store.finish(workspace_id, record)
|
|
201
|
+
return
|
|
202
|
+
request = JobRequest(
|
|
203
|
+
tasks=record.task_set.tasks,
|
|
204
|
+
models=[status.outcomes[i].scenario.model for i in record.service_indices],
|
|
205
|
+
model_seeds=[
|
|
206
|
+
status.outcomes[i].scenario.execution_seed for i in record.service_indices
|
|
207
|
+
],
|
|
208
|
+
submission_id=uuid5(status.id, "simulation"),
|
|
209
|
+
)
|
|
210
|
+
if len(request.model_dump_json().encode()) > self.store.settings.max_body_bytes:
|
|
211
|
+
raise AppError(
|
|
212
|
+
"BODY_TOO_LARGE",
|
|
213
|
+
"Expanded simulation request exceeds the body limit",
|
|
214
|
+
413,
|
|
215
|
+
)
|
|
216
|
+
while status.state not in TERMINAL:
|
|
217
|
+
try:
|
|
218
|
+
if status.job_id is None:
|
|
219
|
+
await self.gateway.ready()
|
|
220
|
+
accepted = await self.gateway.submit(request)
|
|
221
|
+
status.job_id = accepted.job_id
|
|
222
|
+
if status.cancel_requested:
|
|
223
|
+
status.state = "cancelling"
|
|
224
|
+
await self.gateway.cancel(status.job_id)
|
|
225
|
+
upstream = await self.gateway.status(status.job_id)
|
|
226
|
+
page = await self.gateway.outcomes(status.job_id, record.service_cursor)
|
|
227
|
+
for outcome in page.outcomes:
|
|
228
|
+
if outcome.cursor <= record.service_cursor:
|
|
229
|
+
continue
|
|
230
|
+
if outcome.model_index >= len(record.service_indices):
|
|
231
|
+
raise AppError(
|
|
232
|
+
"SERVICE_INVALID_RESPONSE",
|
|
233
|
+
"Service returned an invalid model index",
|
|
234
|
+
502,
|
|
235
|
+
)
|
|
236
|
+
i = record.service_indices[outcome.model_index]
|
|
237
|
+
error = (
|
|
238
|
+
ErrorEnvelope(code=outcome.error.code, message=outcome.error.message)
|
|
239
|
+
if outcome.error
|
|
240
|
+
else None
|
|
241
|
+
)
|
|
242
|
+
self.complete_outcome(record, i, outcome.status, outcome.result, error)
|
|
243
|
+
record.service_cursor = page.next_cursor
|
|
244
|
+
status.error = None
|
|
245
|
+
retry_start = None
|
|
246
|
+
delay = self.store.settings.poll_seconds
|
|
247
|
+
if upstream.status in {
|
|
248
|
+
"completed",
|
|
249
|
+
"completed_with_errors",
|
|
250
|
+
"cancelled",
|
|
251
|
+
}:
|
|
252
|
+
if any(not o.cursor for o in status.outcomes):
|
|
253
|
+
# Status and outcomes are separate reads; a terminal state must
|
|
254
|
+
# have a complete outcome set by the time the page is fetched.
|
|
255
|
+
raise AppError(
|
|
256
|
+
"SERVICE_INVALID_RESPONSE",
|
|
257
|
+
"Terminal job is missing outcomes",
|
|
258
|
+
502,
|
|
259
|
+
)
|
|
260
|
+
status.state = upstream.status
|
|
261
|
+
self.store.finish(workspace_id, record)
|
|
262
|
+
return
|
|
263
|
+
status.state = "cancelling" if status.cancel_requested else upstream.status
|
|
264
|
+
except AppError as exc:
|
|
265
|
+
if exc.code != "SERVICE_UNAVAILABLE":
|
|
266
|
+
raise
|
|
267
|
+
status.state = "reconnecting"
|
|
268
|
+
status.error = ErrorEnvelope(**exc.body())
|
|
269
|
+
retry_start = retry_start or time.monotonic()
|
|
270
|
+
if time.monotonic() - retry_start >= self.store.settings.retry_seconds:
|
|
271
|
+
status.retry_paused = True
|
|
272
|
+
return
|
|
273
|
+
delay = min(5, delay * 2)
|
|
274
|
+
await asyncio.sleep(delay)
|
|
275
|
+
except asyncio.CancelledError:
|
|
276
|
+
raise
|
|
277
|
+
except Exception as exc:
|
|
278
|
+
logger.exception("Application run %s failed", status.id)
|
|
279
|
+
error = (
|
|
280
|
+
exc
|
|
281
|
+
if isinstance(exc, AppError)
|
|
282
|
+
else AppError("INTERNAL_ERROR", "Unexpected application execution error", 500)
|
|
283
|
+
)
|
|
284
|
+
if error.code == "JOB_NOT_FOUND":
|
|
285
|
+
error = AppError(
|
|
286
|
+
"SERVICE_JOB_LOST",
|
|
287
|
+
"The simulation job expired or the service restarted. Completed results were retained.",
|
|
288
|
+
404,
|
|
289
|
+
)
|
|
290
|
+
status.error = ErrorEnvelope(**error.body())
|
|
291
|
+
if status.job_id:
|
|
292
|
+
try:
|
|
293
|
+
await self.gateway.cancel(status.job_id)
|
|
294
|
+
except Exception:
|
|
295
|
+
logger.warning("Unable to confirm cancellation for job %s", status.job_id)
|
|
296
|
+
for i, o in enumerate(status.outcomes):
|
|
297
|
+
if not o.cursor:
|
|
298
|
+
self.complete_outcome(record, i, Status.FAILED, error=status.error)
|
|
299
|
+
status.state = "failed"
|
|
300
|
+
self.store.finish(workspace_id, record)
|
|
301
|
+
|
|
302
|
+
async def cancel_run(self, workspace_id: UUID, run_id: UUID) -> RunStatus:
|
|
303
|
+
"""Cancel a run that has not reached a terminal state.
|
|
304
|
+
|
|
305
|
+
Args:
|
|
306
|
+
workspace_id (UUID): Identifier of the workspace.
|
|
307
|
+
run_id (UUID): Identifier of the simulation run.
|
|
308
|
+
|
|
309
|
+
Returns:
|
|
310
|
+
RunStatus: Updated status of the cancelled run.
|
|
311
|
+
"""
|
|
312
|
+
record = self.store.run(workspace_id, run_id)
|
|
313
|
+
if record.status.state not in TERMINAL:
|
|
314
|
+
record.status.cancel_requested = True
|
|
315
|
+
record.status.state = "cancelling"
|
|
316
|
+
if run_id not in self.tasks:
|
|
317
|
+
self.launch(workspace_id, record)
|
|
318
|
+
return record.status
|
|
319
|
+
|
|
320
|
+
def resume(self, workspace_id: UUID, run_id: UUID) -> RunStatus:
|
|
321
|
+
"""Resume a paused simulation run.
|
|
322
|
+
|
|
323
|
+
Args:
|
|
324
|
+
workspace_id (UUID): Identifier of the workspace.
|
|
325
|
+
run_id (UUID): Identifier of the simulation run.
|
|
326
|
+
|
|
327
|
+
Returns:
|
|
328
|
+
RunStatus: Current status of the resumed run.
|
|
329
|
+
"""
|
|
330
|
+
record = self.store.run(workspace_id, run_id)
|
|
331
|
+
if record.status.state not in TERMINAL and run_id not in self.tasks:
|
|
332
|
+
self.launch(workspace_id, record)
|
|
333
|
+
return record.status
|
|
334
|
+
|
|
335
|
+
async def close(self) -> None:
|
|
336
|
+
"""Close the client and release its resources."""
|
|
337
|
+
self.closed = True
|
|
338
|
+
|
|
339
|
+
async def cancel(record: RunRecord):
|
|
340
|
+
if record.status.job_id:
|
|
341
|
+
try:
|
|
342
|
+
await self.gateway.cancel(record.status.job_id)
|
|
343
|
+
except Exception:
|
|
344
|
+
logger.warning(
|
|
345
|
+
"Shutdown could not confirm cancellation for %s",
|
|
346
|
+
record.status.job_id,
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
try:
|
|
350
|
+
await asyncio.wait_for(
|
|
351
|
+
asyncio.gather(*(cancel(r) for r in self.store.active_records())),
|
|
352
|
+
timeout=5,
|
|
353
|
+
)
|
|
354
|
+
except asyncio.TimeoutError:
|
|
355
|
+
logger.warning("Shutdown cancellation timed out")
|
|
356
|
+
tasks = list(self.tasks.values())
|
|
357
|
+
for task in tasks:
|
|
358
|
+
task.cancel()
|
|
359
|
+
await asyncio.gather(*tasks, return_exceptions=True)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Application errors shared by routes, storage, and the service gateway."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class AppError(Exception):
|
|
5
|
+
"""Represents a structured error returned by the application."""
|
|
6
|
+
|
|
7
|
+
def __init__(self, code: str, message: str, status: int = 422, details=None):
|
|
8
|
+
super().__init__(message)
|
|
9
|
+
self.code, self.message, self.status, self.details = (
|
|
10
|
+
code,
|
|
11
|
+
message,
|
|
12
|
+
status,
|
|
13
|
+
details,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
def body(self):
|
|
17
|
+
return {"code": self.code, "message": self.message, "details": self.details}
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""Streaming CSV exports from immutable scenario snapshots."""
|
|
2
|
+
|
|
3
|
+
from typing import Generator, Literal
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import io
|
|
7
|
+
from .generation import canonical
|
|
8
|
+
from .schemas import ENGINE_VERSION, GENERATOR_VERSION
|
|
9
|
+
from .storage import RunRecord, WorkspaceStore
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def safe_text(value):
|
|
13
|
+
if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@", "\t", "\r")):
|
|
14
|
+
return "'" + value
|
|
15
|
+
return value
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def export_csv(
|
|
19
|
+
store: WorkspaceStore,
|
|
20
|
+
record: RunRecord,
|
|
21
|
+
kind: Literal["summary", "events", "resources"],
|
|
22
|
+
) -> Generator[str, None, None]:
|
|
23
|
+
"""Export a simulation result as CSV.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
store (WorkspaceStore): Store containing the run results.
|
|
27
|
+
record (RunRecord): Run record to export.
|
|
28
|
+
kind (Literal["summary", "events", "resources"]): Data section to export.
|
|
29
|
+
|
|
30
|
+
Yields:
|
|
31
|
+
str: One CSV row at a time.
|
|
32
|
+
"""
|
|
33
|
+
common = [
|
|
34
|
+
"scenario_id",
|
|
35
|
+
"scenario_name",
|
|
36
|
+
"scenario_revision",
|
|
37
|
+
"task_set_id",
|
|
38
|
+
"task_set_hash",
|
|
39
|
+
"run_id",
|
|
40
|
+
"comparison_name",
|
|
41
|
+
"execution_seed",
|
|
42
|
+
"task_generation_seed",
|
|
43
|
+
"team_generation_seed",
|
|
44
|
+
"generator_version",
|
|
45
|
+
"engine_version",
|
|
46
|
+
"status",
|
|
47
|
+
"settings",
|
|
48
|
+
"team_efficiencies",
|
|
49
|
+
]
|
|
50
|
+
fields = {
|
|
51
|
+
"summary": [
|
|
52
|
+
"error_code",
|
|
53
|
+
"error_message",
|
|
54
|
+
"completion_time",
|
|
55
|
+
"initial_value",
|
|
56
|
+
"delivered_value",
|
|
57
|
+
"value_lost_percent",
|
|
58
|
+
],
|
|
59
|
+
"events": [
|
|
60
|
+
"event",
|
|
61
|
+
"event_type",
|
|
62
|
+
"time",
|
|
63
|
+
"status",
|
|
64
|
+
"loss",
|
|
65
|
+
"task_type",
|
|
66
|
+
"is_rework",
|
|
67
|
+
"duration",
|
|
68
|
+
"resource_id",
|
|
69
|
+
"stage_loss_percent",
|
|
70
|
+
],
|
|
71
|
+
"resources": [
|
|
72
|
+
"time",
|
|
73
|
+
"state",
|
|
74
|
+
"waiting",
|
|
75
|
+
"allocated",
|
|
76
|
+
"active",
|
|
77
|
+
"success_t",
|
|
78
|
+
"failure_t",
|
|
79
|
+
"interruption_t",
|
|
80
|
+
"waiting_t",
|
|
81
|
+
"idle_t",
|
|
82
|
+
],
|
|
83
|
+
}[kind]
|
|
84
|
+
# Event status is distinct from the scenario's terminal status.
|
|
85
|
+
headers = ["scenario_status" if x == "status" else x for x in common] + fields
|
|
86
|
+
output = io.StringIO(newline="")
|
|
87
|
+
writer = csv.writer(output)
|
|
88
|
+
writer.writerow(headers)
|
|
89
|
+
yield output.getvalue()
|
|
90
|
+
for outcome in record.status.outcomes:
|
|
91
|
+
s = outcome.scenario
|
|
92
|
+
base = [
|
|
93
|
+
str(s.id),
|
|
94
|
+
safe_text(s.name),
|
|
95
|
+
s.revision,
|
|
96
|
+
str(record.task_set.id),
|
|
97
|
+
record.task_set.content_hash,
|
|
98
|
+
str(record.status.id),
|
|
99
|
+
safe_text(record.status.name),
|
|
100
|
+
s.execution_seed,
|
|
101
|
+
record.task_set.spec.seed,
|
|
102
|
+
s.team_seed,
|
|
103
|
+
GENERATOR_VERSION,
|
|
104
|
+
ENGINE_VERSION,
|
|
105
|
+
outcome.status,
|
|
106
|
+
canonical(s.settings.model_dump()),
|
|
107
|
+
canonical([d.efficiency for d in s.model.developer_team]),
|
|
108
|
+
]
|
|
109
|
+
if kind == "summary":
|
|
110
|
+
rows = [
|
|
111
|
+
[
|
|
112
|
+
outcome.error.code if outcome.error else "",
|
|
113
|
+
safe_text(outcome.error.message) if outcome.error else "",
|
|
114
|
+
outcome.completion_time,
|
|
115
|
+
sum(t.initial_value for t in record.task_set.tasks),
|
|
116
|
+
outcome.delivered_value,
|
|
117
|
+
outcome.loss_percent,
|
|
118
|
+
]
|
|
119
|
+
]
|
|
120
|
+
elif outcome.status == "succeeded":
|
|
121
|
+
result = store.result(record, s.id)
|
|
122
|
+
data = (
|
|
123
|
+
result.metadata.event_metadata
|
|
124
|
+
if kind == "events"
|
|
125
|
+
else result.metadata.resource_metadata
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
def event_rows():
|
|
129
|
+
for item in data:
|
|
130
|
+
values = item.model_dump(mode="json")
|
|
131
|
+
if kind == "events":
|
|
132
|
+
values["stage_loss_percent"] = -100 * item.loss
|
|
133
|
+
yield [values.get(f) for f in fields]
|
|
134
|
+
|
|
135
|
+
rows = event_rows()
|
|
136
|
+
else:
|
|
137
|
+
rows = []
|
|
138
|
+
for row in rows:
|
|
139
|
+
output.seek(0)
|
|
140
|
+
output.truncate(0)
|
|
141
|
+
writer.writerow(base + row)
|
|
142
|
+
yield output.getvalue()
|