agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,2011 @@
|
|
|
1
|
+
"""Local implementation of the hosted ALK sandbox boundary.
|
|
2
|
+
|
|
3
|
+
The platform talks to this HTTP contract in development. Production can replace the
|
|
4
|
+
implementation with a Kubernetes or microVM service without changing platform job APIs.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import asyncio
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from pathlib import PurePosixPath
|
|
16
|
+
import re
|
|
17
|
+
import shutil
|
|
18
|
+
import signal
|
|
19
|
+
import sys
|
|
20
|
+
import time
|
|
21
|
+
from typing import Any
|
|
22
|
+
import uuid
|
|
23
|
+
|
|
24
|
+
from fastapi import (
|
|
25
|
+
Depends,
|
|
26
|
+
FastAPI,
|
|
27
|
+
File,
|
|
28
|
+
Form,
|
|
29
|
+
Header,
|
|
30
|
+
HTTPException,
|
|
31
|
+
UploadFile,
|
|
32
|
+
status,
|
|
33
|
+
)
|
|
34
|
+
from pydantic import BaseModel, Field, SecretStr, model_validator
|
|
35
|
+
|
|
36
|
+
from fi.simulate.runtime.spec import SecretRef
|
|
37
|
+
|
|
38
|
+
from .credentials import (
|
|
39
|
+
CredentialManifest,
|
|
40
|
+
CredentialRequirement,
|
|
41
|
+
RequirementKind,
|
|
42
|
+
RequirementStatus,
|
|
43
|
+
discover_credentials,
|
|
44
|
+
)
|
|
45
|
+
from .executor import GitHubSourceAcquirer, SourceAcquisitionError
|
|
46
|
+
from .github import parse_github_location
|
|
47
|
+
from .job import (
|
|
48
|
+
AgentConnection,
|
|
49
|
+
ExecutionMode,
|
|
50
|
+
HarnessArtifactPolicy,
|
|
51
|
+
HarnessJob,
|
|
52
|
+
HarnessJobStatus,
|
|
53
|
+
HarnessStage,
|
|
54
|
+
RepositorySource,
|
|
55
|
+
SourceKind,
|
|
56
|
+
SourceVisibility,
|
|
57
|
+
)
|
|
58
|
+
from .packaging import PackagingManifest, inspect_packaging
|
|
59
|
+
from .provision import ProvisionError, source_fingerprint, stop
|
|
60
|
+
from .secrets import resolve_worker_secrets, worker_environment
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
_CONTROLLER_TOKEN = uuid.uuid4().hex
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class LocalSandboxRequest(BaseModel):
|
|
67
|
+
source_path: str | None = None
|
|
68
|
+
source_id: str | None = None
|
|
69
|
+
github_repository: str | None = None
|
|
70
|
+
github_ref: str | None = None
|
|
71
|
+
github_commit_sha: str | None = None
|
|
72
|
+
github_visibility: SourceVisibility = SourceVisibility.PUBLIC
|
|
73
|
+
github_installation_id: str | None = None
|
|
74
|
+
scenario_count: int = Field(default=10, ge=1, le=200)
|
|
75
|
+
seed: int | None = None
|
|
76
|
+
agent_name: str | None = None
|
|
77
|
+
connector: str = "auto"
|
|
78
|
+
connector_config: dict[str, Any] = Field(default_factory=dict)
|
|
79
|
+
secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
|
|
80
|
+
environment_values: dict[str, SecretStr] = Field(default_factory=dict)
|
|
81
|
+
platform_run_id: str | None = None
|
|
82
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
83
|
+
|
|
84
|
+
@model_validator(mode="after")
|
|
85
|
+
def _one_source(self) -> "LocalSandboxRequest":
|
|
86
|
+
if (
|
|
87
|
+
sum(
|
|
88
|
+
bool(value)
|
|
89
|
+
for value in (self.source_path, self.source_id, self.github_repository)
|
|
90
|
+
)
|
|
91
|
+
!= 1
|
|
92
|
+
):
|
|
93
|
+
raise ValueError("exactly_one_source_required")
|
|
94
|
+
_validate_environment_values(self.environment_values, self.secret_refs)
|
|
95
|
+
return self
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class SandboxPreflightRequest(BaseModel):
|
|
99
|
+
source_path: str | None = None
|
|
100
|
+
source_id: str | None = None
|
|
101
|
+
github_repository: str | None = None
|
|
102
|
+
github_visibility: SourceVisibility = SourceVisibility.PUBLIC
|
|
103
|
+
github_installation_id: str | None = None
|
|
104
|
+
connector_config: dict[str, Any] = Field(default_factory=dict)
|
|
105
|
+
secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
|
|
106
|
+
environment_values: dict[str, SecretStr] = Field(default_factory=dict)
|
|
107
|
+
|
|
108
|
+
@model_validator(mode="after")
|
|
109
|
+
def _one_source(self) -> "SandboxPreflightRequest":
|
|
110
|
+
if (
|
|
111
|
+
sum(
|
|
112
|
+
bool(value)
|
|
113
|
+
for value in (self.source_path, self.source_id, self.github_repository)
|
|
114
|
+
)
|
|
115
|
+
!= 1
|
|
116
|
+
):
|
|
117
|
+
raise ValueError("exactly_one_source_required")
|
|
118
|
+
_validate_environment_values(self.environment_values, self.secret_refs)
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class SandboxRerunRequest(BaseModel):
|
|
123
|
+
"""Fresh credentials for replaying an already-built harness session.
|
|
124
|
+
|
|
125
|
+
The saved contract, sealed environment bundle and scenarios are reused. Secret values are
|
|
126
|
+
deliberately supplied again (or resolved from fresh references); they are never recovered
|
|
127
|
+
from the completed job's persisted payload.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
|
|
131
|
+
environment_values: dict[str, SecretStr] = Field(default_factory=dict)
|
|
132
|
+
only: list[str] = Field(default_factory=list)
|
|
133
|
+
|
|
134
|
+
@model_validator(mode="after")
|
|
135
|
+
def _valid_environment(self) -> "SandboxRerunRequest":
|
|
136
|
+
_validate_environment_values(self.environment_values, self.secret_refs)
|
|
137
|
+
return self
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
class SandboxPreflightResponse(BaseModel):
|
|
141
|
+
source_kind: SourceKind
|
|
142
|
+
source_label: str
|
|
143
|
+
ready_to_submit: bool
|
|
144
|
+
checkout_required: bool = False
|
|
145
|
+
credentials: CredentialManifest
|
|
146
|
+
packaging: PackagingManifest | None = None
|
|
147
|
+
notes: list[str] = Field(default_factory=list)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class SandboxJobResponse(BaseModel):
|
|
151
|
+
job: HarnessJob
|
|
152
|
+
status: HarnessJobStatus
|
|
153
|
+
events: list[dict[str, Any]] = Field(default_factory=list)
|
|
154
|
+
stage_outputs: list[dict[str, Any]] = Field(default_factory=list)
|
|
155
|
+
artifact_path: str | None = None
|
|
156
|
+
credentials: CredentialManifest | None = None
|
|
157
|
+
stage_outputs: list[dict[str, Any]] = Field(default_factory=list)
|
|
158
|
+
adjustments: list[dict[str, Any]] = Field(default_factory=list)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class SandboxAdjustmentRequest(BaseModel):
|
|
162
|
+
instruction: str = Field(min_length=1, max_length=2_000)
|
|
163
|
+
client_request_id: str | None = Field(default=None, max_length=128)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class UploadedSourceResponse(BaseModel):
|
|
167
|
+
source_id: str
|
|
168
|
+
name: str
|
|
169
|
+
file_count: int
|
|
170
|
+
total_bytes: int
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class UploadedSecretFileResponse(BaseModel):
|
|
174
|
+
"""Opaque handle for a file that is materialized only at the worker boundary."""
|
|
175
|
+
|
|
176
|
+
environment_name: str
|
|
177
|
+
secret_ref: SecretRef
|
|
178
|
+
size: int
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
class LocalSandbox:
|
|
182
|
+
def __init__(
|
|
183
|
+
self,
|
|
184
|
+
root: Path,
|
|
185
|
+
*,
|
|
186
|
+
max_concurrency: int = 2,
|
|
187
|
+
upload_root: Path | None = None,
|
|
188
|
+
secret_file_root: Path | None = None,
|
|
189
|
+
) -> None:
|
|
190
|
+
self.root = root.expanduser().resolve()
|
|
191
|
+
self.jobs_root = self.root / "jobs"
|
|
192
|
+
self.artifacts_root = self.root / "artifacts"
|
|
193
|
+
self.uploads_root = (
|
|
194
|
+
(
|
|
195
|
+
upload_root
|
|
196
|
+
or Path(
|
|
197
|
+
os.getenv("ALK_SANDBOX_UPLOAD_ROOT", str(self.root / "uploads"))
|
|
198
|
+
)
|
|
199
|
+
)
|
|
200
|
+
.expanduser()
|
|
201
|
+
.resolve()
|
|
202
|
+
)
|
|
203
|
+
self.jobs_root.mkdir(parents=True, exist_ok=True)
|
|
204
|
+
self.artifacts_root.mkdir(parents=True, exist_ok=True)
|
|
205
|
+
self.uploads_root.mkdir(parents=True, exist_ok=True)
|
|
206
|
+
self.secret_files_root = (
|
|
207
|
+
(
|
|
208
|
+
secret_file_root
|
|
209
|
+
or Path(
|
|
210
|
+
os.getenv(
|
|
211
|
+
"ALK_SANDBOX_SECRET_FILE_ROOT",
|
|
212
|
+
str(self.root / "secret-files"),
|
|
213
|
+
)
|
|
214
|
+
)
|
|
215
|
+
)
|
|
216
|
+
.expanduser()
|
|
217
|
+
.resolve()
|
|
218
|
+
)
|
|
219
|
+
# A sandbox-service restart orphans every in-flight worker. Secret files therefore have
|
|
220
|
+
# no legitimate consumer after restart and must fail closed instead of lingering on disk.
|
|
221
|
+
shutil.rmtree(self.secret_files_root, ignore_errors=True)
|
|
222
|
+
(self.secret_files_root / "pending").mkdir(parents=True, mode=0o700)
|
|
223
|
+
(self.secret_files_root / "jobs").mkdir(parents=True, mode=0o700)
|
|
224
|
+
self._semaphore = asyncio.Semaphore(max(1, max_concurrency))
|
|
225
|
+
self._processes: dict[str, asyncio.subprocess.Process] = {}
|
|
226
|
+
self._tasks: dict[str, asyncio.Task[None]] = {}
|
|
227
|
+
# Values supplied through the platform's .env flow live only for the lifetime of the
|
|
228
|
+
# job. Persisted job/state artifacts contain the opaque mounted references created below.
|
|
229
|
+
self._ephemeral_secrets: dict[str, dict[str, str]] = {}
|
|
230
|
+
self._ephemeral_secret_file_names: dict[str, set[str]] = {}
|
|
231
|
+
self._recover_orphans()
|
|
232
|
+
|
|
233
|
+
async def upload_secret_file(
|
|
234
|
+
self, uploaded: UploadFile, environment_name: str
|
|
235
|
+
) -> UploadedSecretFileResponse:
|
|
236
|
+
"""Stage one credential file and return a bearer-style opaque reference.
|
|
237
|
+
|
|
238
|
+
Contents never enter a request JSON, job, bundle, event, log or artifact. The upload is
|
|
239
|
+
claimed by exactly one job and removed at that job's terminal boundary.
|
|
240
|
+
"""
|
|
241
|
+
name = str(environment_name).strip()
|
|
242
|
+
if not _ENVIRONMENT_NAME.fullmatch(name):
|
|
243
|
+
raise HTTPException(status_code=400, detail="environment_name is invalid")
|
|
244
|
+
if name in _RUNNER_RESERVED_ENVIRONMENT or name.startswith("ALK_"):
|
|
245
|
+
raise HTTPException(status_code=400, detail="environment_name is reserved")
|
|
246
|
+
self._purge_stale_secret_files()
|
|
247
|
+
identifier = str(uuid.uuid4())
|
|
248
|
+
name_digest = hashlib.sha256(name.encode()).hexdigest()[:12].upper()
|
|
249
|
+
internal_key = (
|
|
250
|
+
f"ALK_SECRET_FILE_{identifier.replace('-', '').upper()}_{name_digest}"
|
|
251
|
+
)
|
|
252
|
+
destination = self.secret_files_root / "pending" / identifier
|
|
253
|
+
total = 0
|
|
254
|
+
try:
|
|
255
|
+
with destination.open("xb") as handle:
|
|
256
|
+
while chunk := await uploaded.read(1024 * 1024):
|
|
257
|
+
total += len(chunk)
|
|
258
|
+
if total > 5 * 1024 * 1024:
|
|
259
|
+
raise HTTPException(
|
|
260
|
+
status_code=413,
|
|
261
|
+
detail="credential file may not exceed 5 MiB",
|
|
262
|
+
)
|
|
263
|
+
handle.write(chunk)
|
|
264
|
+
destination.chmod(0o600)
|
|
265
|
+
if total == 0:
|
|
266
|
+
raise HTTPException(status_code=400, detail="credential file is empty")
|
|
267
|
+
if name == "GOOGLE_APPLICATION_CREDENTIALS":
|
|
268
|
+
try:
|
|
269
|
+
document = json.loads(destination.read_text(encoding="utf-8"))
|
|
270
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc:
|
|
271
|
+
raise HTTPException(
|
|
272
|
+
status_code=400,
|
|
273
|
+
detail="Google application credentials must be a valid JSON object",
|
|
274
|
+
) from exc
|
|
275
|
+
if not isinstance(document, dict) or not document:
|
|
276
|
+
raise HTTPException(
|
|
277
|
+
status_code=400,
|
|
278
|
+
detail="Google application credentials must be a valid JSON object",
|
|
279
|
+
)
|
|
280
|
+
except Exception:
|
|
281
|
+
destination.unlink(missing_ok=True)
|
|
282
|
+
raise
|
|
283
|
+
return UploadedSecretFileResponse(
|
|
284
|
+
environment_name=name,
|
|
285
|
+
secret_ref=SecretRef(
|
|
286
|
+
manager="mounted",
|
|
287
|
+
key=internal_key,
|
|
288
|
+
purpose=f"job-scoped credential file for {name}",
|
|
289
|
+
),
|
|
290
|
+
size=total,
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
def _claim_secret_files(
|
|
294
|
+
self, job_id: str, references: dict[str, SecretRef]
|
|
295
|
+
) -> tuple[dict[str, str], set[str]]:
|
|
296
|
+
"""Atomically move provider uploads into one job-private directory."""
|
|
297
|
+
self._purge_stale_secret_files()
|
|
298
|
+
candidates: list[tuple[str, str, Path]] = []
|
|
299
|
+
for environment_name, reference in references.items():
|
|
300
|
+
if reference.manager != "mounted" or not reference.key.startswith(
|
|
301
|
+
"ALK_SECRET_FILE_"
|
|
302
|
+
):
|
|
303
|
+
continue
|
|
304
|
+
suffix = reference.key.removeprefix("ALK_SECRET_FILE_")
|
|
305
|
+
match = re.fullmatch(r"([0-9A-F]{32})_([0-9A-F]{12})", suffix)
|
|
306
|
+
expected_digest = (
|
|
307
|
+
hashlib.sha256(environment_name.encode()).hexdigest()[:12].upper()
|
|
308
|
+
)
|
|
309
|
+
if match is None or match.group(2) != expected_digest:
|
|
310
|
+
raise HTTPException(
|
|
311
|
+
status_code=400, detail="secret_file_reference_invalid"
|
|
312
|
+
)
|
|
313
|
+
identifier = str(uuid.UUID(hex=match.group(1).lower()))
|
|
314
|
+
source = self.secret_files_root / "pending" / identifier
|
|
315
|
+
if not source.is_file():
|
|
316
|
+
raise HTTPException(
|
|
317
|
+
status_code=400,
|
|
318
|
+
detail="secret_file_reference_unavailable_or_consumed",
|
|
319
|
+
)
|
|
320
|
+
candidates.append((environment_name, reference.key, source))
|
|
321
|
+
if not candidates:
|
|
322
|
+
return {}, set()
|
|
323
|
+
job_root = self.secret_files_root / "jobs" / job_id
|
|
324
|
+
job_root.mkdir(mode=0o700)
|
|
325
|
+
claimed: dict[str, str] = {}
|
|
326
|
+
names: set[str] = set()
|
|
327
|
+
try:
|
|
328
|
+
for environment_name, internal_key, source in candidates:
|
|
329
|
+
target = job_root / environment_name
|
|
330
|
+
source.replace(target)
|
|
331
|
+
target.chmod(0o400)
|
|
332
|
+
claimed[internal_key] = str(target)
|
|
333
|
+
names.add(environment_name)
|
|
334
|
+
except Exception:
|
|
335
|
+
shutil.rmtree(job_root, ignore_errors=True)
|
|
336
|
+
raise
|
|
337
|
+
return claimed, names
|
|
338
|
+
|
|
339
|
+
def _delete_job_secret_files(self, job_id: str) -> None:
|
|
340
|
+
shutil.rmtree(self.secret_files_root / "jobs" / job_id, ignore_errors=True)
|
|
341
|
+
|
|
342
|
+
def _purge_stale_secret_files(self) -> None:
|
|
343
|
+
ttl = max(60, int(os.getenv("ALK_SECRET_FILE_TTL_SECONDS", "900")))
|
|
344
|
+
cutoff = time.time() - ttl
|
|
345
|
+
for path in (self.secret_files_root / "pending").iterdir():
|
|
346
|
+
try:
|
|
347
|
+
if path.is_file() and path.stat().st_mtime < cutoff:
|
|
348
|
+
path.unlink()
|
|
349
|
+
except FileNotFoundError:
|
|
350
|
+
continue
|
|
351
|
+
|
|
352
|
+
async def upload_source(
|
|
353
|
+
self, files: list[UploadFile], paths: list[str], name: str
|
|
354
|
+
) -> UploadedSourceResponse:
|
|
355
|
+
if not files or len(files) != len(paths):
|
|
356
|
+
raise HTTPException(
|
|
357
|
+
status_code=400, detail="one relative path is required per file"
|
|
358
|
+
)
|
|
359
|
+
if len(files) > 5_000:
|
|
360
|
+
raise HTTPException(
|
|
361
|
+
status_code=413, detail="source may contain at most 5000 files"
|
|
362
|
+
)
|
|
363
|
+
source_id = str(uuid.uuid4())
|
|
364
|
+
staging = self.uploads_root / f".{source_id}.uploading"
|
|
365
|
+
destination = self.uploads_root / source_id
|
|
366
|
+
staging.mkdir(mode=0o700)
|
|
367
|
+
total = 0
|
|
368
|
+
seen: set[str] = set()
|
|
369
|
+
try:
|
|
370
|
+
for uploaded, raw_path in zip(files, paths, strict=True):
|
|
371
|
+
relative = _safe_uploaded_path(raw_path)
|
|
372
|
+
key = relative.as_posix()
|
|
373
|
+
if key in seen:
|
|
374
|
+
raise HTTPException(
|
|
375
|
+
status_code=400, detail=f"duplicate source path: {key}"
|
|
376
|
+
)
|
|
377
|
+
seen.add(key)
|
|
378
|
+
target = staging.joinpath(*relative.parts)
|
|
379
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
380
|
+
file_size = 0
|
|
381
|
+
starts_with_shebang = False
|
|
382
|
+
with target.open("xb") as handle:
|
|
383
|
+
while chunk := await uploaded.read(1024 * 1024):
|
|
384
|
+
if file_size == 0:
|
|
385
|
+
starts_with_shebang = chunk.startswith(b"#!")
|
|
386
|
+
file_size += len(chunk)
|
|
387
|
+
total += len(chunk)
|
|
388
|
+
if file_size > 50 * 1024 * 1024 or total > 200 * 1024 * 1024:
|
|
389
|
+
raise HTTPException(
|
|
390
|
+
status_code=413,
|
|
391
|
+
detail="source exceeds the 50 MiB file or 200 MiB bundle limit",
|
|
392
|
+
)
|
|
393
|
+
handle.write(chunk)
|
|
394
|
+
target.chmod(0o755 if starts_with_shebang else 0o644)
|
|
395
|
+
manifest = {
|
|
396
|
+
"source_id": source_id,
|
|
397
|
+
"name": (name or "uploaded-agent")[:255],
|
|
398
|
+
"file_count": len(files),
|
|
399
|
+
"total_bytes": total,
|
|
400
|
+
"created_at": _now(),
|
|
401
|
+
}
|
|
402
|
+
staging.replace(destination)
|
|
403
|
+
_write_json(
|
|
404
|
+
self.uploads_root / f"{source_id}.json",
|
|
405
|
+
manifest,
|
|
406
|
+
)
|
|
407
|
+
except Exception:
|
|
408
|
+
shutil.rmtree(staging, ignore_errors=True)
|
|
409
|
+
shutil.rmtree(destination, ignore_errors=True)
|
|
410
|
+
(self.uploads_root / f"{source_id}.json").unlink(missing_ok=True)
|
|
411
|
+
raise
|
|
412
|
+
return UploadedSourceResponse(
|
|
413
|
+
source_id=source_id,
|
|
414
|
+
name=(name or "uploaded-agent")[:255],
|
|
415
|
+
file_count=len(files),
|
|
416
|
+
total_bytes=total,
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
def uploaded_source(self, source_id: str) -> Path:
|
|
420
|
+
if not re.fullmatch(r"[0-9a-f-]{36}", source_id):
|
|
421
|
+
raise HTTPException(status_code=400, detail="source_id is invalid")
|
|
422
|
+
source = (self.uploads_root / source_id).resolve()
|
|
423
|
+
if source.parent != self.uploads_root or not source.is_dir():
|
|
424
|
+
raise HTTPException(status_code=404, detail="uploaded source was not found")
|
|
425
|
+
if not (self.uploads_root / f"{source_id}.json").is_file():
|
|
426
|
+
raise HTTPException(status_code=400, detail="uploaded source is incomplete")
|
|
427
|
+
return source
|
|
428
|
+
|
|
429
|
+
def _recover_orphans(self) -> None:
|
|
430
|
+
for path in self.jobs_root.glob("*/state.json"):
|
|
431
|
+
state = _read_json(path)
|
|
432
|
+
if state.get("stage") not in _TERMINAL_STAGES:
|
|
433
|
+
controller_pid = int(state.get("controller_pid", 0) or 0)
|
|
434
|
+
controller_identity = str(state.get("controller_identity") or "")
|
|
435
|
+
controller_token = str(state.get("controller_token") or "")
|
|
436
|
+
if (
|
|
437
|
+
controller_pid == os.getpid()
|
|
438
|
+
and controller_token == _CONTROLLER_TOKEN
|
|
439
|
+
) or (
|
|
440
|
+
controller_pid > 0
|
|
441
|
+
and controller_identity
|
|
442
|
+
and _process_identity(controller_pid) == controller_identity
|
|
443
|
+
):
|
|
444
|
+
# Another app object or health process may inspect the same provider root.
|
|
445
|
+
# It must not declare a job orphaned while the owning controller still exists.
|
|
446
|
+
continue
|
|
447
|
+
state.update(
|
|
448
|
+
stage=HarnessStage.FAILED.value,
|
|
449
|
+
detail="sandbox service restarted while the job was running",
|
|
450
|
+
updated_at=_now(),
|
|
451
|
+
)
|
|
452
|
+
_write_json(path, state)
|
|
453
|
+
|
|
454
|
+
def submit(self, request: LocalSandboxRequest) -> SandboxJobResponse:
|
|
455
|
+
source = (
|
|
456
|
+
_allowed_source(request.source_path)
|
|
457
|
+
if request.source_path
|
|
458
|
+
else self.uploaded_source(request.source_id)
|
|
459
|
+
if request.source_id
|
|
460
|
+
else None
|
|
461
|
+
)
|
|
462
|
+
identifier = str(uuid.uuid4())
|
|
463
|
+
run_id = f"harness-{identifier}"
|
|
464
|
+
github_location = (
|
|
465
|
+
_github_location(request.github_repository)
|
|
466
|
+
if request.github_repository
|
|
467
|
+
else None
|
|
468
|
+
)
|
|
469
|
+
github_repository = github_location.repository if github_location else None
|
|
470
|
+
mounted_refs: dict[str, SecretRef] = {}
|
|
471
|
+
mounted_values: dict[str, str] = {}
|
|
472
|
+
for index, (name, value) in enumerate(request.environment_values.items()):
|
|
473
|
+
internal_key = f"ALK_JOB_{identifier.replace('-', '').upper()}_{index}"
|
|
474
|
+
mounted_refs[name] = SecretRef(
|
|
475
|
+
manager="mounted",
|
|
476
|
+
key=internal_key,
|
|
477
|
+
purpose=f"job-scoped environment value for {name}",
|
|
478
|
+
)
|
|
479
|
+
mounted_values[internal_key] = value.get_secret_value()
|
|
480
|
+
claimed_files, secret_file_names = self._claim_secret_files(
|
|
481
|
+
identifier, request.secret_refs
|
|
482
|
+
)
|
|
483
|
+
mounted_values.update(claimed_files)
|
|
484
|
+
try:
|
|
485
|
+
source_spec = (
|
|
486
|
+
RepositorySource(
|
|
487
|
+
kind=SourceKind.ARCHIVE,
|
|
488
|
+
archive_artifact_id=request.source_id,
|
|
489
|
+
)
|
|
490
|
+
if request.source_id
|
|
491
|
+
else RepositorySource(
|
|
492
|
+
kind=SourceKind.LOCAL_REPOSITORY, local_path=str(source)
|
|
493
|
+
)
|
|
494
|
+
if source
|
|
495
|
+
else RepositorySource(
|
|
496
|
+
kind=SourceKind.GITHUB,
|
|
497
|
+
repository=github_repository,
|
|
498
|
+
ref=(github_location.ref if github_location else None)
|
|
499
|
+
or request.github_ref,
|
|
500
|
+
commit_sha=request.github_commit_sha,
|
|
501
|
+
visibility=request.github_visibility,
|
|
502
|
+
installation_id=request.github_installation_id,
|
|
503
|
+
)
|
|
504
|
+
)
|
|
505
|
+
job = HarnessJob(
|
|
506
|
+
job_id=identifier,
|
|
507
|
+
run_id=run_id,
|
|
508
|
+
execution=(
|
|
509
|
+
ExecutionMode.HOSTED
|
|
510
|
+
if request.source_id or request.github_repository
|
|
511
|
+
else ExecutionMode.LOCAL
|
|
512
|
+
),
|
|
513
|
+
source=source_spec,
|
|
514
|
+
agent=AgentConnection(
|
|
515
|
+
connector=request.connector,
|
|
516
|
+
config=request.connector_config,
|
|
517
|
+
secret_refs={**request.secret_refs, **mounted_refs},
|
|
518
|
+
),
|
|
519
|
+
scenario_count=request.scenario_count,
|
|
520
|
+
seed=request.seed,
|
|
521
|
+
artifacts=HarnessArtifactPolicy(
|
|
522
|
+
level="full", allow_bundle_download=True
|
|
523
|
+
),
|
|
524
|
+
platform_run_id=request.platform_run_id,
|
|
525
|
+
metadata={
|
|
526
|
+
**request.metadata,
|
|
527
|
+
"agent_name": request.agent_name
|
|
528
|
+
or (
|
|
529
|
+
_uploaded_source_name(source)
|
|
530
|
+
if request.source_id and source
|
|
531
|
+
else source.name
|
|
532
|
+
if source
|
|
533
|
+
else str(github_repository).split("/")[-1]
|
|
534
|
+
),
|
|
535
|
+
"source_kind": source_spec.kind.value,
|
|
536
|
+
# Both dotenv values and credential-file paths are runtime configuration. Only
|
|
537
|
+
# their names are persisted; values and provider-local paths remain ephemeral.
|
|
538
|
+
"environment_value_names": sorted(
|
|
539
|
+
{*mounted_refs, *secret_file_names}
|
|
540
|
+
),
|
|
541
|
+
"secret_file_names": sorted(secret_file_names),
|
|
542
|
+
},
|
|
543
|
+
)
|
|
544
|
+
directory = self.jobs_root / identifier
|
|
545
|
+
directory.mkdir()
|
|
546
|
+
_write_json(directory / "job.json", job.model_dump(mode="json"))
|
|
547
|
+
_write_json(
|
|
548
|
+
directory / "state.json",
|
|
549
|
+
{
|
|
550
|
+
"job_id": identifier,
|
|
551
|
+
"run_id": run_id,
|
|
552
|
+
"stage": HarnessStage.QUEUED.value,
|
|
553
|
+
"updated_at": _now(),
|
|
554
|
+
"detail": "waiting for a local sandbox slot",
|
|
555
|
+
"completed_scenarios": 0,
|
|
556
|
+
"total_scenarios": request.scenario_count,
|
|
557
|
+
"attempt": 1,
|
|
558
|
+
"controller_pid": os.getpid(),
|
|
559
|
+
"controller_identity": _process_identity(os.getpid()),
|
|
560
|
+
"controller_token": _CONTROLLER_TOKEN,
|
|
561
|
+
},
|
|
562
|
+
)
|
|
563
|
+
except Exception:
|
|
564
|
+
self._delete_job_secret_files(identifier)
|
|
565
|
+
shutil.rmtree(self.jobs_root / identifier, ignore_errors=True)
|
|
566
|
+
raise
|
|
567
|
+
if mounted_values:
|
|
568
|
+
self._ephemeral_secrets[identifier] = mounted_values
|
|
569
|
+
if secret_file_names:
|
|
570
|
+
self._ephemeral_secret_file_names[identifier] = secret_file_names
|
|
571
|
+
self._tasks[identifier] = asyncio.create_task(self._execute(job, source))
|
|
572
|
+
return self.get(identifier)
|
|
573
|
+
|
|
574
|
+
def rerun(self, job_id: str, request: SandboxRerunRequest) -> SandboxJobResponse:
|
|
575
|
+
response = self.get(job_id)
|
|
576
|
+
if not response.status.stage.terminal:
|
|
577
|
+
raise HTTPException(
|
|
578
|
+
status_code=409, detail="harness job is already running"
|
|
579
|
+
)
|
|
580
|
+
output = self.artifacts_root / response.job.run_id
|
|
581
|
+
required = (
|
|
582
|
+
output / "contract.json",
|
|
583
|
+
output / "scenarios.json",
|
|
584
|
+
output / "environment-bundle" / "manifest.json",
|
|
585
|
+
output / "platform.json",
|
|
586
|
+
)
|
|
587
|
+
missing = [path.name for path in required if not path.is_file()]
|
|
588
|
+
if missing:
|
|
589
|
+
raise HTTPException(
|
|
590
|
+
status_code=409,
|
|
591
|
+
detail=(
|
|
592
|
+
"saved harness session cannot be rerun; missing "
|
|
593
|
+
+ ", ".join(missing)
|
|
594
|
+
),
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
runtime_configuration = {
|
|
598
|
+
name: value.get_secret_value()
|
|
599
|
+
for name, value in request.environment_values.items()
|
|
600
|
+
}
|
|
601
|
+
claimed_files, secret_file_names = self._claim_secret_files(
|
|
602
|
+
job_id, request.secret_refs
|
|
603
|
+
)
|
|
604
|
+
try:
|
|
605
|
+
resolved = resolve_worker_secrets(
|
|
606
|
+
request.secret_refs,
|
|
607
|
+
environment={**os.environ, **claimed_files},
|
|
608
|
+
)
|
|
609
|
+
except Exception:
|
|
610
|
+
self._delete_job_secret_files(job_id)
|
|
611
|
+
raise
|
|
612
|
+
runtime_configuration.update(resolved)
|
|
613
|
+
self._ephemeral_secrets[job_id] = dict(runtime_configuration)
|
|
614
|
+
self._ephemeral_secret_file_names[job_id] = secret_file_names
|
|
615
|
+
|
|
616
|
+
state_path = self.jobs_root / job_id / "state.json"
|
|
617
|
+
state = _read_json(state_path)
|
|
618
|
+
operation_started_at = _now()
|
|
619
|
+
state.update(
|
|
620
|
+
stage=HarnessStage.QUEUED.value,
|
|
621
|
+
detail="waiting to restart the saved environment",
|
|
622
|
+
failure=None,
|
|
623
|
+
completed_scenarios=0,
|
|
624
|
+
total_scenarios=(len(request.only) or response.job.scenario_count),
|
|
625
|
+
operation="rerun",
|
|
626
|
+
operation_started_at=operation_started_at,
|
|
627
|
+
attempt=int(state.get("attempt", 0) or 0) + 1,
|
|
628
|
+
updated_at=operation_started_at,
|
|
629
|
+
)
|
|
630
|
+
_write_json(state_path, state)
|
|
631
|
+
self._tasks[job_id] = asyncio.create_task(
|
|
632
|
+
self._execute_rerun(response.job, request.only)
|
|
633
|
+
)
|
|
634
|
+
return self.get(job_id)
|
|
635
|
+
|
|
636
|
+
async def _execute_rerun(self, job: HarnessJob, only: list[str]) -> None:
|
|
637
|
+
"""Run calls from immutable saved artifacts, with a fresh environment lifecycle."""
|
|
638
|
+
|
|
639
|
+
directory = self.jobs_root / job.job_id
|
|
640
|
+
state_path = directory / "state.json"
|
|
641
|
+
output = self.artifacts_root / job.run_id
|
|
642
|
+
# The local runner controls a host Docker daemon. Paths in Compose bind mounts are
|
|
643
|
+
# therefore resolved by that daemon, not inside this service container. Artifacts live
|
|
644
|
+
# on the runner's private volume, so replaying a bundle directly from ``output`` makes
|
|
645
|
+
# repository seed/config files invisible to Docker. Materialize a disposable copy in
|
|
646
|
+
# the provider's Docker-visible upload workspace. Hosted providers must offer the same
|
|
647
|
+
# workspace guarantee even when their implementation is not a nested Docker daemon.
|
|
648
|
+
replay_root = self.uploads_root / ".reruns" / job.job_id
|
|
649
|
+
async with self._semaphore:
|
|
650
|
+
state = _read_json(state_path)
|
|
651
|
+
if state.get("stage") == HarnessStage.CANCELED.value:
|
|
652
|
+
return
|
|
653
|
+
state.update(
|
|
654
|
+
stage=HarnessStage.RUNNING.value,
|
|
655
|
+
detail="restarting the saved environment and running scenarios",
|
|
656
|
+
updated_at=_now(),
|
|
657
|
+
)
|
|
658
|
+
_write_json(state_path, state)
|
|
659
|
+
log_handle = (directory / "worker.log").open("ab")
|
|
660
|
+
try:
|
|
661
|
+
shutil.rmtree(replay_root, ignore_errors=True)
|
|
662
|
+
replay_root.parent.mkdir(parents=True, exist_ok=True)
|
|
663
|
+
shutil.copytree(output / "environment-bundle", replay_root)
|
|
664
|
+
runtime_configuration = self._ephemeral_secrets.get(job.job_id, {})
|
|
665
|
+
child_environment = worker_environment(
|
|
666
|
+
{}, runtime_configuration=runtime_configuration
|
|
667
|
+
)
|
|
668
|
+
child_environment["ALK_RUNTIME_CONFIGURATION_NAMES"] = ",".join(
|
|
669
|
+
sorted(runtime_configuration)
|
|
670
|
+
)
|
|
671
|
+
child_environment["ALK_RUNTIME_SECRET_FILE_NAMES"] = ",".join(
|
|
672
|
+
sorted(self._ephemeral_secret_file_names.get(job.job_id, set()))
|
|
673
|
+
)
|
|
674
|
+
child_environment["ALK_HOSTED_EXECUTION"] = (
|
|
675
|
+
"1" if job.execution is ExecutionMode.HOSTED else "0"
|
|
676
|
+
)
|
|
677
|
+
child_environment["ALK_HARNESS_JOB_ID"] = job.job_id
|
|
678
|
+
|
|
679
|
+
async def run_stage(command: list[str]) -> int:
|
|
680
|
+
process = await asyncio.create_subprocess_exec(
|
|
681
|
+
*command,
|
|
682
|
+
stdout=log_handle,
|
|
683
|
+
stderr=asyncio.subprocess.STDOUT,
|
|
684
|
+
start_new_session=True,
|
|
685
|
+
env=child_environment,
|
|
686
|
+
)
|
|
687
|
+
self._processes[job.job_id] = process
|
|
688
|
+
return await process.wait()
|
|
689
|
+
|
|
690
|
+
environment_up = [
|
|
691
|
+
sys.executable,
|
|
692
|
+
"-m",
|
|
693
|
+
"fi.alk.harness.cli",
|
|
694
|
+
"environment",
|
|
695
|
+
"up",
|
|
696
|
+
"--bundle",
|
|
697
|
+
str(replay_root),
|
|
698
|
+
"--out",
|
|
699
|
+
str(output),
|
|
700
|
+
]
|
|
701
|
+
simulation = [
|
|
702
|
+
sys.executable,
|
|
703
|
+
"-m",
|
|
704
|
+
"fi.alk.harness.cli",
|
|
705
|
+
"simulate",
|
|
706
|
+
"--name",
|
|
707
|
+
str(job.metadata.get("agent_name") or job.run_id),
|
|
708
|
+
"--out",
|
|
709
|
+
str(output),
|
|
710
|
+
]
|
|
711
|
+
if only:
|
|
712
|
+
simulation.extend(["--only", *only])
|
|
713
|
+
environment_down = [
|
|
714
|
+
sys.executable,
|
|
715
|
+
"-m",
|
|
716
|
+
"fi.alk.harness.cli",
|
|
717
|
+
"environment",
|
|
718
|
+
"down",
|
|
719
|
+
"--out",
|
|
720
|
+
str(output),
|
|
721
|
+
]
|
|
722
|
+
|
|
723
|
+
up_code = await run_stage(environment_up)
|
|
724
|
+
simulation_code = 1
|
|
725
|
+
cleanup_code = 0
|
|
726
|
+
if up_code == 0:
|
|
727
|
+
try:
|
|
728
|
+
simulation_code = await run_stage(simulation)
|
|
729
|
+
finally:
|
|
730
|
+
cleanup_code = await run_stage(environment_down)
|
|
731
|
+
|
|
732
|
+
# Exit 2 is a completed suite whose submitted agent failed checks.
|
|
733
|
+
successful_execution = (
|
|
734
|
+
up_code == 0 and simulation_code in (0, 2) and cleanup_code == 0
|
|
735
|
+
)
|
|
736
|
+
failed_stage = (
|
|
737
|
+
"environment_up"
|
|
738
|
+
if up_code != 0
|
|
739
|
+
else "environment_down"
|
|
740
|
+
if cleanup_code != 0
|
|
741
|
+
else "simulation"
|
|
742
|
+
)
|
|
743
|
+
failure_stage = (
|
|
744
|
+
HarnessStage.BUILDING_ENVIRONMENT.value
|
|
745
|
+
if up_code != 0
|
|
746
|
+
else HarnessStage.CLEANING_UP.value
|
|
747
|
+
if cleanup_code != 0
|
|
748
|
+
else HarnessStage.RUNNING.value
|
|
749
|
+
)
|
|
750
|
+
return_code = (
|
|
751
|
+
up_code
|
|
752
|
+
if up_code != 0
|
|
753
|
+
else cleanup_code
|
|
754
|
+
if cleanup_code != 0
|
|
755
|
+
else simulation_code
|
|
756
|
+
)
|
|
757
|
+
state = _read_json(state_path)
|
|
758
|
+
if state.get("stage") != HarnessStage.CANCELED.value:
|
|
759
|
+
state.update(
|
|
760
|
+
stage=(
|
|
761
|
+
HarnessStage.COMPLETED.value
|
|
762
|
+
if successful_execution
|
|
763
|
+
else HarnessStage.FAILED.value
|
|
764
|
+
),
|
|
765
|
+
detail=(
|
|
766
|
+
"saved harness session rerun completed"
|
|
767
|
+
if successful_execution
|
|
768
|
+
else f"saved harness session rerun exited {return_code}"
|
|
769
|
+
),
|
|
770
|
+
completed_scenarios=(len(only) if only else job.scenario_count)
|
|
771
|
+
if successful_execution
|
|
772
|
+
else 0,
|
|
773
|
+
failure=None
|
|
774
|
+
if successful_execution
|
|
775
|
+
else {
|
|
776
|
+
"domain": "environment",
|
|
777
|
+
"stage": failure_stage,
|
|
778
|
+
"code": "saved_session_rerun_failed",
|
|
779
|
+
"message": f"{failed_stage} exited {return_code}",
|
|
780
|
+
"retryable": False,
|
|
781
|
+
},
|
|
782
|
+
updated_at=_now(),
|
|
783
|
+
)
|
|
784
|
+
_write_json(state_path, state)
|
|
785
|
+
except Exception as exc:
|
|
786
|
+
state = _read_json(state_path)
|
|
787
|
+
state.update(
|
|
788
|
+
stage=HarnessStage.FAILED.value,
|
|
789
|
+
detail=f"{type(exc).__name__}: {exc}",
|
|
790
|
+
failure={
|
|
791
|
+
"domain": "infrastructure",
|
|
792
|
+
"stage": HarnessStage.RUNNING.value,
|
|
793
|
+
"code": type(exc).__name__,
|
|
794
|
+
"message": str(exc),
|
|
795
|
+
"retryable": False,
|
|
796
|
+
},
|
|
797
|
+
updated_at=_now(),
|
|
798
|
+
)
|
|
799
|
+
_write_json(state_path, state)
|
|
800
|
+
finally:
|
|
801
|
+
log_handle.close()
|
|
802
|
+
self._processes.pop(job.job_id, None)
|
|
803
|
+
self._tasks.pop(job.job_id, None)
|
|
804
|
+
self._ephemeral_secrets.pop(job.job_id, None)
|
|
805
|
+
self._ephemeral_secret_file_names.pop(job.job_id, None)
|
|
806
|
+
self._delete_job_secret_files(job.job_id)
|
|
807
|
+
shutil.rmtree(replay_root, ignore_errors=True)
|
|
808
|
+
|
|
809
|
+
def preflight(self, request: SandboxPreflightRequest) -> SandboxPreflightResponse:
|
|
810
|
+
if request.source_path or request.source_id:
|
|
811
|
+
source = (
|
|
812
|
+
_allowed_source(request.source_path)
|
|
813
|
+
if request.source_path
|
|
814
|
+
else self.uploaded_source(request.source_id or "")
|
|
815
|
+
)
|
|
816
|
+
packaging = inspect_packaging(
|
|
817
|
+
source,
|
|
818
|
+
external_environment=bool(request.environment_values),
|
|
819
|
+
)
|
|
820
|
+
manifest = discover_credentials(
|
|
821
|
+
source,
|
|
822
|
+
secret_refs=request.secret_refs,
|
|
823
|
+
provided_environment={
|
|
824
|
+
**request.connector_config,
|
|
825
|
+
**{
|
|
826
|
+
name: value.get_secret_value()
|
|
827
|
+
for name, value in request.environment_values.items()
|
|
828
|
+
},
|
|
829
|
+
},
|
|
830
|
+
scan_paths=_credential_scan_paths(packaging),
|
|
831
|
+
)
|
|
832
|
+
return SandboxPreflightResponse(
|
|
833
|
+
source_kind=(
|
|
834
|
+
SourceKind.ARCHIVE
|
|
835
|
+
if request.source_id
|
|
836
|
+
else SourceKind.LOCAL_REPOSITORY
|
|
837
|
+
),
|
|
838
|
+
source_label=(
|
|
839
|
+
_uploaded_source_name(source) if request.source_id else str(source)
|
|
840
|
+
),
|
|
841
|
+
ready_to_submit=(
|
|
842
|
+
manifest.ready
|
|
843
|
+
and packaging.ready
|
|
844
|
+
and (packaging.agent_runtime_packaged or not packaging.candidates)
|
|
845
|
+
),
|
|
846
|
+
credentials=manifest,
|
|
847
|
+
packaging=packaging,
|
|
848
|
+
)
|
|
849
|
+
location = _github_location(request.github_repository or "")
|
|
850
|
+
repository = location.repository
|
|
851
|
+
requirement = CredentialRequirement(
|
|
852
|
+
id="github_installation",
|
|
853
|
+
environment_name="GITHUB_INSTALLATION",
|
|
854
|
+
provider="github",
|
|
855
|
+
purpose="read private repository source",
|
|
856
|
+
kind=RequirementKind.SECRET,
|
|
857
|
+
required=request.github_visibility is SourceVisibility.PRIVATE,
|
|
858
|
+
status=(
|
|
859
|
+
RequirementStatus.CONFIGURED
|
|
860
|
+
if request.github_installation_id
|
|
861
|
+
else RequirementStatus.MISSING
|
|
862
|
+
if request.github_visibility is SourceVisibility.PRIVATE
|
|
863
|
+
else RequirementStatus.OPTIONAL
|
|
864
|
+
),
|
|
865
|
+
accepted_secret_types=["github_app_installation"],
|
|
866
|
+
)
|
|
867
|
+
manifest = CredentialManifest(
|
|
868
|
+
source_digest="0" * 64,
|
|
869
|
+
detected_connectors=[],
|
|
870
|
+
requirements=[requirement],
|
|
871
|
+
scanned_files=0,
|
|
872
|
+
)
|
|
873
|
+
return SandboxPreflightResponse(
|
|
874
|
+
source_kind=SourceKind.GITHUB,
|
|
875
|
+
source_label=repository,
|
|
876
|
+
ready_to_submit=manifest.ready,
|
|
877
|
+
checkout_required=True,
|
|
878
|
+
credentials=manifest,
|
|
879
|
+
notes=[
|
|
880
|
+
"Agent credential discovery continues inside the sandbox after checkout."
|
|
881
|
+
],
|
|
882
|
+
)
|
|
883
|
+
|
|
884
|
+
async def _execute(self, job: HarnessJob, source: Path | None) -> None:
|
|
885
|
+
directory = self.jobs_root / job.job_id
|
|
886
|
+
state_path = directory / "state.json"
|
|
887
|
+
output = self.artifacts_root / job.run_id
|
|
888
|
+
async with self._semaphore:
|
|
889
|
+
state = _read_json(state_path)
|
|
890
|
+
if state.get("stage") == HarnessStage.CANCELED.value:
|
|
891
|
+
return
|
|
892
|
+
state.update(
|
|
893
|
+
stage=HarnessStage.ACQUIRING_SOURCE.value,
|
|
894
|
+
detail="acquiring source inside the ALK runner",
|
|
895
|
+
updated_at=_now(),
|
|
896
|
+
)
|
|
897
|
+
_write_json(state_path, state)
|
|
898
|
+
log_handle = (directory / "worker.log").open("ab")
|
|
899
|
+
try:
|
|
900
|
+
if source is None:
|
|
901
|
+
workspace = directory / "workspace"
|
|
902
|
+
for attempt in range(1, job.retry.max_infrastructure_attempts + 1):
|
|
903
|
+
state.update(
|
|
904
|
+
attempt=attempt,
|
|
905
|
+
detail=f"cloning public GitHub source (attempt {attempt})",
|
|
906
|
+
updated_at=_now(),
|
|
907
|
+
)
|
|
908
|
+
_write_json(state_path, state)
|
|
909
|
+
try:
|
|
910
|
+
source = await GitHubSourceAcquirer(
|
|
911
|
+
_github_installation_token
|
|
912
|
+
).acquire(job, workspace)
|
|
913
|
+
break
|
|
914
|
+
except SourceAcquisitionError:
|
|
915
|
+
checkout = workspace / "repository"
|
|
916
|
+
if checkout.exists():
|
|
917
|
+
shutil.rmtree(checkout)
|
|
918
|
+
if attempt >= job.retry.max_infrastructure_attempts:
|
|
919
|
+
raise
|
|
920
|
+
delay = min(
|
|
921
|
+
job.retry.max_backoff_seconds,
|
|
922
|
+
job.retry.initial_backoff_seconds
|
|
923
|
+
* (2 ** (attempt - 1)),
|
|
924
|
+
)
|
|
925
|
+
await asyncio.sleep(delay)
|
|
926
|
+
if source is None:
|
|
927
|
+
raise SourceAcquisitionError("github_checkout_missing")
|
|
928
|
+
packaging_manifest = inspect_packaging(
|
|
929
|
+
source,
|
|
930
|
+
external_environment=bool(
|
|
931
|
+
job.metadata.get("environment_value_names", [])
|
|
932
|
+
),
|
|
933
|
+
)
|
|
934
|
+
credential_manifest = discover_credentials(
|
|
935
|
+
source,
|
|
936
|
+
secret_refs=job.agent.secret_refs,
|
|
937
|
+
provided_environment=job.agent.config,
|
|
938
|
+
scan_paths=_credential_scan_paths(packaging_manifest),
|
|
939
|
+
)
|
|
940
|
+
_write_json(
|
|
941
|
+
directory / "packaging.json",
|
|
942
|
+
packaging_manifest.model_dump(mode="json"),
|
|
943
|
+
)
|
|
944
|
+
if not packaging_manifest.ready:
|
|
945
|
+
detail = (
|
|
946
|
+
"; ".join(packaging_manifest.notes)
|
|
947
|
+
or "packaging is not runnable"
|
|
948
|
+
)
|
|
949
|
+
state.update(
|
|
950
|
+
stage=HarnessStage.FAILED.value,
|
|
951
|
+
detail=f"repository packaging preflight failed: {detail}",
|
|
952
|
+
failure={
|
|
953
|
+
"domain": "environment",
|
|
954
|
+
"stage": HarnessStage.ACQUIRING_SOURCE.value,
|
|
955
|
+
"code": "packaging_preflight_failed",
|
|
956
|
+
"message": detail,
|
|
957
|
+
"retryable": False,
|
|
958
|
+
},
|
|
959
|
+
updated_at=_now(),
|
|
960
|
+
)
|
|
961
|
+
_write_json(state_path, state)
|
|
962
|
+
return
|
|
963
|
+
_write_json(
|
|
964
|
+
directory / "credentials.json",
|
|
965
|
+
credential_manifest.model_dump(mode="json"),
|
|
966
|
+
)
|
|
967
|
+
if not credential_manifest.ready:
|
|
968
|
+
missing = [
|
|
969
|
+
item.environment_name
|
|
970
|
+
for item in credential_manifest.missing_required
|
|
971
|
+
]
|
|
972
|
+
missing.extend(
|
|
973
|
+
f"one of {choice.options}"
|
|
974
|
+
for choice in credential_manifest.credential_choices
|
|
975
|
+
if not choice.satisfied
|
|
976
|
+
)
|
|
977
|
+
names = ", ".join(missing)
|
|
978
|
+
state.update(
|
|
979
|
+
stage=HarnessStage.FAILED.value,
|
|
980
|
+
detail=f"missing required credentials: {names}",
|
|
981
|
+
failure={
|
|
982
|
+
"domain": "connectivity",
|
|
983
|
+
"stage": HarnessStage.ACQUIRING_SOURCE.value,
|
|
984
|
+
"code": "credentials_missing",
|
|
985
|
+
"message": f"Configure secret references for: {names}",
|
|
986
|
+
"retryable": False,
|
|
987
|
+
},
|
|
988
|
+
updated_at=_now(),
|
|
989
|
+
)
|
|
990
|
+
_write_json(state_path, state)
|
|
991
|
+
return
|
|
992
|
+
resolved_secrets = resolve_worker_secrets(
|
|
993
|
+
job.agent.secret_refs,
|
|
994
|
+
environment={
|
|
995
|
+
**os.environ,
|
|
996
|
+
**self._ephemeral_secrets.get(job.job_id, {}),
|
|
997
|
+
},
|
|
998
|
+
)
|
|
999
|
+
submitted_source_digest = source_fingerprint(source)
|
|
1000
|
+
result: dict[str, Any] = {}
|
|
1001
|
+
return_code = 1
|
|
1002
|
+
for worker_attempt in range(
|
|
1003
|
+
1, job.retry.max_infrastructure_attempts + 1
|
|
1004
|
+
):
|
|
1005
|
+
state.update(
|
|
1006
|
+
attempt=worker_attempt,
|
|
1007
|
+
detail=f"running isolated harness worker (attempt {worker_attempt})",
|
|
1008
|
+
updated_at=_now(),
|
|
1009
|
+
)
|
|
1010
|
+
_write_json(state_path, state)
|
|
1011
|
+
if worker_attempt > 1 and output.exists():
|
|
1012
|
+
archived = directory / f"failed-attempt-{worker_attempt - 1}"
|
|
1013
|
+
if archived.exists():
|
|
1014
|
+
shutil.rmtree(archived)
|
|
1015
|
+
output.replace(archived)
|
|
1016
|
+
runtime_names = {
|
|
1017
|
+
str(name)
|
|
1018
|
+
for name in job.metadata.get("environment_value_names", [])
|
|
1019
|
+
}
|
|
1020
|
+
runtime_configuration = {
|
|
1021
|
+
name: value
|
|
1022
|
+
for name, value in resolved_secrets.items()
|
|
1023
|
+
if name in runtime_names
|
|
1024
|
+
}
|
|
1025
|
+
controller_configuration = {
|
|
1026
|
+
**_configuration_environment(job.agent.config),
|
|
1027
|
+
**{
|
|
1028
|
+
name: value
|
|
1029
|
+
for name, value in resolved_secrets.items()
|
|
1030
|
+
if name not in runtime_names
|
|
1031
|
+
},
|
|
1032
|
+
}
|
|
1033
|
+
child_environment = worker_environment(
|
|
1034
|
+
controller_configuration,
|
|
1035
|
+
runtime_configuration=runtime_configuration,
|
|
1036
|
+
)
|
|
1037
|
+
# The provisioner must explicitly pass customer-provided values into
|
|
1038
|
+
# Dockerfile/generated runtime containers. Persist names only; values stay
|
|
1039
|
+
# in this child process environment and never enter the job or bundle.
|
|
1040
|
+
child_environment["ALK_RUNTIME_CONFIGURATION_NAMES"] = ",".join(
|
|
1041
|
+
sorted(runtime_configuration)
|
|
1042
|
+
)
|
|
1043
|
+
child_environment["ALK_RUNTIME_SECRET_FILE_NAMES"] = ",".join(
|
|
1044
|
+
sorted(self._ephemeral_secret_file_names.get(job.job_id, set()))
|
|
1045
|
+
)
|
|
1046
|
+
child_environment["ALK_HOSTED_EXECUTION"] = (
|
|
1047
|
+
"1" if job.execution is ExecutionMode.HOSTED else "0"
|
|
1048
|
+
)
|
|
1049
|
+
child_environment["ALK_HARNESS_JOB_ID"] = job.job_id
|
|
1050
|
+
process = await asyncio.create_subprocess_exec(
|
|
1051
|
+
sys.executable,
|
|
1052
|
+
"-m",
|
|
1053
|
+
"fi.alk.harness.sandbox_worker",
|
|
1054
|
+
str(directory / "job.json"),
|
|
1055
|
+
"--source",
|
|
1056
|
+
str(source),
|
|
1057
|
+
"--output",
|
|
1058
|
+
str(output),
|
|
1059
|
+
"--status",
|
|
1060
|
+
str(directory / "result.json"),
|
|
1061
|
+
"--adjustments",
|
|
1062
|
+
str(directory / "adjustments.jsonl"),
|
|
1063
|
+
stdout=log_handle,
|
|
1064
|
+
stderr=asyncio.subprocess.STDOUT,
|
|
1065
|
+
start_new_session=True,
|
|
1066
|
+
env=child_environment,
|
|
1067
|
+
)
|
|
1068
|
+
self._processes[job.job_id] = process
|
|
1069
|
+
return_code = await process.wait()
|
|
1070
|
+
result = _read_json(directory / "result.json")
|
|
1071
|
+
if source_fingerprint(source) != submitted_source_digest:
|
|
1072
|
+
return_code = 1
|
|
1073
|
+
result = {
|
|
1074
|
+
"stage": HarnessStage.FAILED.value,
|
|
1075
|
+
"detail": "submitted source changed during isolated execution",
|
|
1076
|
+
"failure": {
|
|
1077
|
+
"domain": "infrastructure",
|
|
1078
|
+
"stage": HarnessStage.CLEANING_UP.value,
|
|
1079
|
+
"code": "source_mutation_detected",
|
|
1080
|
+
"message": (
|
|
1081
|
+
"The runner detected writes to the read-only source tree"
|
|
1082
|
+
),
|
|
1083
|
+
"retryable": False,
|
|
1084
|
+
},
|
|
1085
|
+
}
|
|
1086
|
+
failure = result.get("failure") or {}
|
|
1087
|
+
retryable = _worker_failure_retryable(
|
|
1088
|
+
return_code,
|
|
1089
|
+
failure,
|
|
1090
|
+
job.retry.retryable_domains,
|
|
1091
|
+
)
|
|
1092
|
+
if (
|
|
1093
|
+
not retryable
|
|
1094
|
+
or worker_attempt >= job.retry.max_infrastructure_attempts
|
|
1095
|
+
):
|
|
1096
|
+
break
|
|
1097
|
+
delay = min(
|
|
1098
|
+
job.retry.max_backoff_seconds,
|
|
1099
|
+
job.retry.initial_backoff_seconds * (2 ** (worker_attempt - 1)),
|
|
1100
|
+
)
|
|
1101
|
+
state.update(
|
|
1102
|
+
detail=(
|
|
1103
|
+
f"retrying {failure.get('domain', 'infrastructure')} "
|
|
1104
|
+
f"failure in {delay:g}s"
|
|
1105
|
+
),
|
|
1106
|
+
updated_at=_now(),
|
|
1107
|
+
)
|
|
1108
|
+
_write_json(state_path, state)
|
|
1109
|
+
await asyncio.sleep(delay)
|
|
1110
|
+
state = _read_json(state_path)
|
|
1111
|
+
if state.get("stage") != HarnessStage.CANCELED.value:
|
|
1112
|
+
state.update(
|
|
1113
|
+
stage=result.get(
|
|
1114
|
+
"stage",
|
|
1115
|
+
HarnessStage.COMPLETED.value
|
|
1116
|
+
if return_code == 0
|
|
1117
|
+
else HarnessStage.FAILED.value,
|
|
1118
|
+
),
|
|
1119
|
+
detail=result.get("detail")
|
|
1120
|
+
or (
|
|
1121
|
+
None if return_code == 0 else f"worker exited {return_code}"
|
|
1122
|
+
),
|
|
1123
|
+
completed_scenarios=result.get("completed_scenarios", 0),
|
|
1124
|
+
failure=result.get("failure"),
|
|
1125
|
+
attempt=state.get("attempt", 1),
|
|
1126
|
+
updated_at=_now(),
|
|
1127
|
+
)
|
|
1128
|
+
_write_json(state_path, state)
|
|
1129
|
+
except Exception as exc:
|
|
1130
|
+
state = _read_json(state_path)
|
|
1131
|
+
domain = (
|
|
1132
|
+
"connectivity"
|
|
1133
|
+
if isinstance(exc, SourceAcquisitionError)
|
|
1134
|
+
else "infrastructure"
|
|
1135
|
+
)
|
|
1136
|
+
state.update(
|
|
1137
|
+
stage=HarnessStage.FAILED.value,
|
|
1138
|
+
detail=f"{type(exc).__name__}: {exc}",
|
|
1139
|
+
failure={
|
|
1140
|
+
"domain": domain,
|
|
1141
|
+
"stage": HarnessStage.ACQUIRING_SOURCE.value,
|
|
1142
|
+
"code": type(exc).__name__,
|
|
1143
|
+
"message": str(exc),
|
|
1144
|
+
"retryable": isinstance(exc, SourceAcquisitionError),
|
|
1145
|
+
},
|
|
1146
|
+
updated_at=_now(),
|
|
1147
|
+
)
|
|
1148
|
+
_write_json(state_path, state)
|
|
1149
|
+
finally:
|
|
1150
|
+
log_handle.close()
|
|
1151
|
+
self._processes.pop(job.job_id, None)
|
|
1152
|
+
self._tasks.pop(job.job_id, None)
|
|
1153
|
+
self._ephemeral_secrets.pop(job.job_id, None)
|
|
1154
|
+
self._ephemeral_secret_file_names.pop(job.job_id, None)
|
|
1155
|
+
self._delete_job_secret_files(job.job_id)
|
|
1156
|
+
|
|
1157
|
+
def get(self, job_id: str) -> SandboxJobResponse:
|
|
1158
|
+
directory = self.jobs_root / job_id
|
|
1159
|
+
if not directory.is_dir():
|
|
1160
|
+
raise KeyError(job_id)
|
|
1161
|
+
job = HarnessJob.model_validate(_read_json(directory / "job.json"))
|
|
1162
|
+
raw_state = _read_json(directory / "state.json")
|
|
1163
|
+
events = _events(self.artifacts_root / job.run_id / "harness-events.jsonl")
|
|
1164
|
+
event_stage = _stage_from_events(events)
|
|
1165
|
+
if (
|
|
1166
|
+
raw_state.get("operation") == "rerun"
|
|
1167
|
+
and raw_state.get("stage") not in _TERMINAL_STAGES
|
|
1168
|
+
):
|
|
1169
|
+
# Historic event streams end in ``completed``. During a rerun the current state is
|
|
1170
|
+
# authoritative until the new invocation emits/commits its terminal result.
|
|
1171
|
+
event_stage = None
|
|
1172
|
+
stage = event_stage or raw_state.get("stage", "queued")
|
|
1173
|
+
updated_at = raw_state.get("updated_at", _now())
|
|
1174
|
+
detail = raw_state.get("detail")
|
|
1175
|
+
if event_stage and events:
|
|
1176
|
+
# While the worker is alive, state.json is intentionally written only
|
|
1177
|
+
# at process boundaries. The canonical event stream is the live
|
|
1178
|
+
# source of truth, so expose its timestamp and stage instead of making
|
|
1179
|
+
# a healthy long-running job look stale in the platform UI.
|
|
1180
|
+
updated_at = events[-1].get("wall_time") or updated_at
|
|
1181
|
+
detail = {
|
|
1182
|
+
HarnessStage.UNDERSTANDING_AGENT.value: "understanding agent source",
|
|
1183
|
+
HarnessStage.GENERATING_ENVIRONMENT.value: "provisioning and validating environment",
|
|
1184
|
+
HarnessStage.GENERATING_SCENARIOS.value: "generating and validating scenarios",
|
|
1185
|
+
HarnessStage.RUNNING.value: "running scenarios",
|
|
1186
|
+
}.get(event_stage, detail)
|
|
1187
|
+
if raw_state.get("stage") in _TERMINAL_STAGES:
|
|
1188
|
+
stage = raw_state["stage"]
|
|
1189
|
+
updated_at = raw_state.get("updated_at", updated_at)
|
|
1190
|
+
detail = raw_state.get("detail")
|
|
1191
|
+
scenario_index = _json_artifact(
|
|
1192
|
+
self.artifacts_root / job.run_id / "scenarios.json"
|
|
1193
|
+
)
|
|
1194
|
+
discovered_scenarios = (
|
|
1195
|
+
len(scenario_index) if isinstance(scenario_index, list) else 0
|
|
1196
|
+
)
|
|
1197
|
+
completed_scenarios = int(raw_state.get("completed_scenarios", 0) or 0)
|
|
1198
|
+
if stage == HarnessStage.RUNNING.value:
|
|
1199
|
+
# The worker writes state.json only at process boundaries, while each scenario result
|
|
1200
|
+
# is committed as soon as that scenario finishes. Derive live progress from the newest
|
|
1201
|
+
# campaign directory so a multi-call run does not remain at 0/N until finalization.
|
|
1202
|
+
# Only immediate scenario result files count; nested SDK/debug artifacts are ignored.
|
|
1203
|
+
runs_root = self.artifacts_root / job.run_id / "runs"
|
|
1204
|
+
campaigns = [path for path in runs_root.glob("run-*") if path.is_dir()]
|
|
1205
|
+
operation_started_at = str(
|
|
1206
|
+
raw_state.get("operation_started_at") or ""
|
|
1207
|
+
).strip()
|
|
1208
|
+
if raw_state.get("operation") == "rerun" and operation_started_at:
|
|
1209
|
+
try:
|
|
1210
|
+
started_ns = int(
|
|
1211
|
+
datetime.fromisoformat(operation_started_at).timestamp()
|
|
1212
|
+
* 1_000_000_000
|
|
1213
|
+
)
|
|
1214
|
+
campaigns = [
|
|
1215
|
+
path
|
|
1216
|
+
for path in campaigns
|
|
1217
|
+
if path.stat().st_mtime_ns >= started_ns
|
|
1218
|
+
]
|
|
1219
|
+
except ValueError:
|
|
1220
|
+
campaigns = []
|
|
1221
|
+
if campaigns:
|
|
1222
|
+
latest = max(campaigns, key=lambda path: path.stat().st_mtime_ns)
|
|
1223
|
+
committed = sum(
|
|
1224
|
+
1
|
|
1225
|
+
for scenario in latest.iterdir()
|
|
1226
|
+
if scenario.is_dir() and (scenario / "result.json").is_file()
|
|
1227
|
+
)
|
|
1228
|
+
completed_scenarios = min(
|
|
1229
|
+
raw_state.get("total_scenarios", job.scenario_count),
|
|
1230
|
+
max(completed_scenarios, committed),
|
|
1231
|
+
)
|
|
1232
|
+
failure = raw_state.get("failure")
|
|
1233
|
+
if isinstance(failure, dict):
|
|
1234
|
+
legacy_stage = str(failure.get("stage") or "")
|
|
1235
|
+
normalized_stage = {
|
|
1236
|
+
"environment_up": HarnessStage.BUILDING_ENVIRONMENT.value,
|
|
1237
|
+
"environment_down": HarnessStage.CLEANING_UP.value,
|
|
1238
|
+
"simulation": HarnessStage.RUNNING.value,
|
|
1239
|
+
}.get(legacy_stage)
|
|
1240
|
+
if normalized_stage:
|
|
1241
|
+
failure = {**failure, "stage": normalized_stage}
|
|
1242
|
+
status_value = HarnessJobStatus(
|
|
1243
|
+
job_id=job.job_id,
|
|
1244
|
+
run_id=job.run_id,
|
|
1245
|
+
stage=stage,
|
|
1246
|
+
updated_at=updated_at,
|
|
1247
|
+
detail=detail,
|
|
1248
|
+
failure=failure,
|
|
1249
|
+
completed_scenarios=completed_scenarios,
|
|
1250
|
+
total_scenarios=max(
|
|
1251
|
+
raw_state.get("total_scenarios", job.scenario_count),
|
|
1252
|
+
discovered_scenarios,
|
|
1253
|
+
),
|
|
1254
|
+
attempt=raw_state.get("attempt", 1),
|
|
1255
|
+
)
|
|
1256
|
+
return SandboxJobResponse(
|
|
1257
|
+
job=job,
|
|
1258
|
+
status=status_value,
|
|
1259
|
+
events=events,
|
|
1260
|
+
stage_outputs=_stage_outputs(self.artifacts_root / job.run_id),
|
|
1261
|
+
artifact_path=str(self.artifacts_root / job.run_id),
|
|
1262
|
+
credentials=(
|
|
1263
|
+
CredentialManifest.model_validate(
|
|
1264
|
+
_read_json(directory / "credentials.json")
|
|
1265
|
+
)
|
|
1266
|
+
if (directory / "credentials.json").is_file()
|
|
1267
|
+
else None
|
|
1268
|
+
),
|
|
1269
|
+
adjustments=_adjustments(directory),
|
|
1270
|
+
)
|
|
1271
|
+
|
|
1272
|
+
def list(self) -> list[SandboxJobResponse]:
|
|
1273
|
+
responses = []
|
|
1274
|
+
for directory in sorted(
|
|
1275
|
+
self.jobs_root.iterdir(),
|
|
1276
|
+
key=lambda path: path.stat().st_mtime,
|
|
1277
|
+
reverse=True,
|
|
1278
|
+
):
|
|
1279
|
+
if directory.is_dir():
|
|
1280
|
+
responses.append(self.get(directory.name))
|
|
1281
|
+
return responses
|
|
1282
|
+
|
|
1283
|
+
async def cancel(self, job_id: str) -> SandboxJobResponse:
|
|
1284
|
+
response = self.get(job_id)
|
|
1285
|
+
if response.status.stage.terminal:
|
|
1286
|
+
return response
|
|
1287
|
+
state_path = self.jobs_root / job_id / "state.json"
|
|
1288
|
+
state = _read_json(state_path)
|
|
1289
|
+
state.update(
|
|
1290
|
+
stage=HarnessStage.CANCELED.value,
|
|
1291
|
+
detail="canceled by user",
|
|
1292
|
+
updated_at=_now(),
|
|
1293
|
+
)
|
|
1294
|
+
_write_json(state_path, state)
|
|
1295
|
+
process = self._processes.get(job_id)
|
|
1296
|
+
if process and process.returncode is None:
|
|
1297
|
+
try:
|
|
1298
|
+
os.killpg(process.pid, signal.SIGTERM)
|
|
1299
|
+
except ProcessLookupError:
|
|
1300
|
+
pass
|
|
1301
|
+
try:
|
|
1302
|
+
await asyncio.wait_for(process.wait(), timeout=10)
|
|
1303
|
+
except TimeoutError:
|
|
1304
|
+
try:
|
|
1305
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
1306
|
+
except ProcessLookupError:
|
|
1307
|
+
pass
|
|
1308
|
+
task = self._tasks.get(job_id)
|
|
1309
|
+
if task and not task.done():
|
|
1310
|
+
task.cancel()
|
|
1311
|
+
# SIGTERM stops the worker before its async ``finally`` can reliably run. Cancellation
|
|
1312
|
+
# must therefore own the same environment boundary explicitly; otherwise every timed-out
|
|
1313
|
+
# hosted job leaves its database, network and volumes behind.
|
|
1314
|
+
output = self.artifacts_root / response.job.run_id
|
|
1315
|
+
try:
|
|
1316
|
+
await asyncio.to_thread(stop, output)
|
|
1317
|
+
except ProvisionError as exc:
|
|
1318
|
+
state.update(
|
|
1319
|
+
detail=f"canceled; isolated environment cleanup failed: {exc}",
|
|
1320
|
+
updated_at=_now(),
|
|
1321
|
+
)
|
|
1322
|
+
_write_json(state_path, state)
|
|
1323
|
+
self._ephemeral_secrets.pop(job_id, None)
|
|
1324
|
+
self._ephemeral_secret_file_names.pop(job_id, None)
|
|
1325
|
+
self._delete_job_secret_files(job_id)
|
|
1326
|
+
return self.get(job_id)
|
|
1327
|
+
|
|
1328
|
+
def adjust(
|
|
1329
|
+
self, job_id: str, request: SandboxAdjustmentRequest
|
|
1330
|
+
) -> SandboxJobResponse:
|
|
1331
|
+
response = self.get(job_id)
|
|
1332
|
+
if response.status.stage.terminal:
|
|
1333
|
+
raise HTTPException(
|
|
1334
|
+
status_code=409, detail="a completed run cannot be adjusted"
|
|
1335
|
+
)
|
|
1336
|
+
if response.status.stage in {
|
|
1337
|
+
HarnessStage.CLEANING_UP,
|
|
1338
|
+
HarnessStage.UPLOADING_ARTIFACTS,
|
|
1339
|
+
}:
|
|
1340
|
+
raise HTTPException(
|
|
1341
|
+
status_code=409,
|
|
1342
|
+
detail="the run is already finalizing; start a follow-up run to change it",
|
|
1343
|
+
)
|
|
1344
|
+
instruction = request.instruction.strip()
|
|
1345
|
+
record = {
|
|
1346
|
+
"adjustment_id": str(uuid.uuid4()),
|
|
1347
|
+
"client_request_id": request.client_request_id,
|
|
1348
|
+
"instruction": instruction,
|
|
1349
|
+
"target_stage": _adjustment_stage(instruction, response.status.stage.value),
|
|
1350
|
+
"scenario_delta": _scenario_delta(instruction),
|
|
1351
|
+
"status": "pending",
|
|
1352
|
+
"created_at": _now(),
|
|
1353
|
+
}
|
|
1354
|
+
path = self.jobs_root / job_id / "adjustments.jsonl"
|
|
1355
|
+
with path.open("a", encoding="utf-8") as stream:
|
|
1356
|
+
stream.write(json.dumps(record, separators=(",", ":")) + "\n")
|
|
1357
|
+
stream.flush()
|
|
1358
|
+
return self.get(job_id)
|
|
1359
|
+
|
|
1360
|
+
|
|
1361
|
+
def _configuration_environment(config: dict[str, Any]) -> dict[str, str]:
|
|
1362
|
+
"""Convert explicit, non-secret connector configuration into child env vars."""
|
|
1363
|
+
result: dict[str, str] = {}
|
|
1364
|
+
for name, value in config.items():
|
|
1365
|
+
if not re.fullmatch(r"[A-Z][A-Z0-9_]{2,}", str(name)):
|
|
1366
|
+
continue
|
|
1367
|
+
if isinstance(value, bool):
|
|
1368
|
+
result[str(name)] = "true" if value else "false"
|
|
1369
|
+
elif isinstance(value, (str, int, float)):
|
|
1370
|
+
result[str(name)] = str(value)
|
|
1371
|
+
return result
|
|
1372
|
+
|
|
1373
|
+
|
|
1374
|
+
_ENVIRONMENT_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
1375
|
+
_RUNNER_RESERVED_ENVIRONMENT = {
|
|
1376
|
+
"DOCKER_HOST",
|
|
1377
|
+
"FI_API_KEY",
|
|
1378
|
+
"FI_BASE_URL",
|
|
1379
|
+
"FI_SECRET_KEY",
|
|
1380
|
+
"HARNESS_PLATFORM_API_KEY",
|
|
1381
|
+
"HARNESS_PLATFORM_SECRET_KEY",
|
|
1382
|
+
"HARNESS_PLATFORM_URL",
|
|
1383
|
+
"HARNESS_WEBHOOK_HOST",
|
|
1384
|
+
"HARNESS_WEBHOOK_PORT",
|
|
1385
|
+
"HOME",
|
|
1386
|
+
"PATH",
|
|
1387
|
+
"PYTHONPATH",
|
|
1388
|
+
}
|
|
1389
|
+
|
|
1390
|
+
|
|
1391
|
+
def _validate_environment_values(
|
|
1392
|
+
values: dict[str, SecretStr], references: dict[str, SecretRef]
|
|
1393
|
+
) -> None:
|
|
1394
|
+
if len(values) > 256:
|
|
1395
|
+
raise ValueError("environment_values_limit_exceeded")
|
|
1396
|
+
if set(values) & set(references):
|
|
1397
|
+
raise ValueError("environment_value_conflicts_with_secret_reference")
|
|
1398
|
+
total = 0
|
|
1399
|
+
for name, secret in values.items():
|
|
1400
|
+
if not _ENVIRONMENT_NAME.fullmatch(name):
|
|
1401
|
+
raise ValueError(f"environment_name_invalid: {name}")
|
|
1402
|
+
if name in _RUNNER_RESERVED_ENVIRONMENT or name.startswith("ALK_"):
|
|
1403
|
+
raise ValueError(f"environment_name_reserved: {name}")
|
|
1404
|
+
value = secret.get_secret_value()
|
|
1405
|
+
if "\x00" in value:
|
|
1406
|
+
raise ValueError(f"environment_value_contains_nul: {name}")
|
|
1407
|
+
total += len(name.encode()) + len(value.encode())
|
|
1408
|
+
if total > 262_144:
|
|
1409
|
+
raise ValueError("environment_values_size_exceeded")
|
|
1410
|
+
|
|
1411
|
+
|
|
1412
|
+
def _credential_scan_paths(packaging: PackagingManifest) -> list[str] | None:
|
|
1413
|
+
"""Scope preflight credentials to the Compose runtime that will actually run."""
|
|
1414
|
+
if packaging.selected_kind is None or packaging.selected_kind.value != "compose":
|
|
1415
|
+
return None
|
|
1416
|
+
selected = next(
|
|
1417
|
+
(
|
|
1418
|
+
candidate
|
|
1419
|
+
for candidate in packaging.candidates
|
|
1420
|
+
if candidate.path == packaging.selected_path
|
|
1421
|
+
),
|
|
1422
|
+
None,
|
|
1423
|
+
)
|
|
1424
|
+
if selected is None or packaging.selected_path is None:
|
|
1425
|
+
return None
|
|
1426
|
+
return [packaging.selected_path, *selected.runtime_source_roots]
|
|
1427
|
+
|
|
1428
|
+
|
|
1429
|
+
def _worker_failure_retryable(
|
|
1430
|
+
return_code: int, failure: dict[str, Any], retryable_domains: list[str]
|
|
1431
|
+
) -> bool:
|
|
1432
|
+
"""Retry infrastructure transport, never agent behavior or grading outcomes."""
|
|
1433
|
+
return (
|
|
1434
|
+
return_code != 0
|
|
1435
|
+
and failure.get("retryable") is True
|
|
1436
|
+
and failure.get("domain") in retryable_domains
|
|
1437
|
+
)
|
|
1438
|
+
|
|
1439
|
+
|
|
1440
|
+
def _allowed_source(raw: str) -> Path:
|
|
1441
|
+
source_text = raw
|
|
1442
|
+
for mapping in os.getenv("ALK_SANDBOX_PATH_MAP", "").split(os.pathsep):
|
|
1443
|
+
if "=" not in mapping:
|
|
1444
|
+
continue
|
|
1445
|
+
external, internal = mapping.split("=", 1)
|
|
1446
|
+
if source_text == external or source_text.startswith(
|
|
1447
|
+
external.rstrip("/") + "/"
|
|
1448
|
+
):
|
|
1449
|
+
source_text = (
|
|
1450
|
+
internal.rstrip("/") + source_text[len(external.rstrip("/")) :]
|
|
1451
|
+
)
|
|
1452
|
+
break
|
|
1453
|
+
source = Path(source_text).expanduser().resolve()
|
|
1454
|
+
if not source.is_dir():
|
|
1455
|
+
raise HTTPException(
|
|
1456
|
+
status_code=400, detail="source_path must be an existing directory"
|
|
1457
|
+
)
|
|
1458
|
+
configured = os.getenv("ALK_SANDBOX_SOURCE_ROOTS")
|
|
1459
|
+
roots = [
|
|
1460
|
+
Path(item).expanduser().resolve()
|
|
1461
|
+
for item in (
|
|
1462
|
+
configured.split(os.pathsep) if configured else [str(Path.cwd().parent)]
|
|
1463
|
+
)
|
|
1464
|
+
if item
|
|
1465
|
+
]
|
|
1466
|
+
if not any(source == root or root in source.parents for root in roots):
|
|
1467
|
+
raise HTTPException(
|
|
1468
|
+
status_code=403, detail="source_path is outside allowed roots"
|
|
1469
|
+
)
|
|
1470
|
+
return source
|
|
1471
|
+
|
|
1472
|
+
|
|
1473
|
+
def _safe_uploaded_path(raw: str) -> PurePosixPath:
|
|
1474
|
+
normalized = str(raw).replace("\\", "/").strip("/")
|
|
1475
|
+
relative = PurePosixPath(normalized)
|
|
1476
|
+
if (
|
|
1477
|
+
not normalized
|
|
1478
|
+
or len(normalized) > 1024
|
|
1479
|
+
or relative.is_absolute()
|
|
1480
|
+
or any(part in {"", ".", ".."} for part in relative.parts)
|
|
1481
|
+
or len(relative.parts) > 64
|
|
1482
|
+
):
|
|
1483
|
+
raise HTTPException(status_code=400, detail=f"unsafe source path: {raw[:200]}")
|
|
1484
|
+
leaf = relative.name.lower()
|
|
1485
|
+
safe_environment_templates = {".env.example", ".env.sample", ".env.template"}
|
|
1486
|
+
if (
|
|
1487
|
+
leaf == ".env" or leaf.startswith(".env.")
|
|
1488
|
+
) and leaf not in safe_environment_templates:
|
|
1489
|
+
raise HTTPException(
|
|
1490
|
+
status_code=400,
|
|
1491
|
+
detail=f"{relative.as_posix()} may contain secrets; upload it through the .env control",
|
|
1492
|
+
)
|
|
1493
|
+
return relative
|
|
1494
|
+
|
|
1495
|
+
|
|
1496
|
+
def _uploaded_source_name(source: Path) -> str:
|
|
1497
|
+
manifest = _read_json(source.parent / f"{source.name}.json")
|
|
1498
|
+
return str(manifest.get("name") or "uploaded-agent")
|
|
1499
|
+
|
|
1500
|
+
|
|
1501
|
+
def _github_location(raw: str):
|
|
1502
|
+
try:
|
|
1503
|
+
return parse_github_location(raw)
|
|
1504
|
+
except ValueError as exc:
|
|
1505
|
+
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
|
1506
|
+
|
|
1507
|
+
|
|
1508
|
+
def _github_installation_token(installation_id: str) -> str:
|
|
1509
|
+
"""Local broker adapter; production exchanges the installation via its vault."""
|
|
1510
|
+
safe_id = re.sub(r"[^A-Za-z0-9_]", "_", installation_id)
|
|
1511
|
+
token = os.getenv(f"GITHUB_INSTALLATION_{safe_id}_TOKEN")
|
|
1512
|
+
if not token:
|
|
1513
|
+
raise SourceAcquisitionError(
|
|
1514
|
+
"github_installation_token_unavailable; authorize the GitHub App installation"
|
|
1515
|
+
)
|
|
1516
|
+
return token
|
|
1517
|
+
|
|
1518
|
+
|
|
1519
|
+
def _events(path: Path) -> list[dict[str, Any]]:
|
|
1520
|
+
if not path.exists():
|
|
1521
|
+
return []
|
|
1522
|
+
result = []
|
|
1523
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
1524
|
+
try:
|
|
1525
|
+
result.append(json.loads(line))
|
|
1526
|
+
except ValueError:
|
|
1527
|
+
continue
|
|
1528
|
+
return result
|
|
1529
|
+
|
|
1530
|
+
|
|
1531
|
+
def _json_artifact(path: Path, *, max_bytes: int = 1_000_000) -> Any | None:
|
|
1532
|
+
"""Read a bounded, generated JSON artifact for control-plane presentation."""
|
|
1533
|
+
try:
|
|
1534
|
+
if not path.is_file() or path.stat().st_size > max_bytes:
|
|
1535
|
+
return None
|
|
1536
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
1537
|
+
except (OSError, ValueError):
|
|
1538
|
+
return None
|
|
1539
|
+
|
|
1540
|
+
|
|
1541
|
+
def _stage_outputs(root: Path) -> list[dict[str, Any]]:
|
|
1542
|
+
outputs: list[dict[str, Any]] = []
|
|
1543
|
+
contract = _json_artifact(root / "contract.json")
|
|
1544
|
+
if isinstance(contract, dict):
|
|
1545
|
+
outputs.append(
|
|
1546
|
+
{
|
|
1547
|
+
"id": "contract",
|
|
1548
|
+
"stage": HarnessStage.UNDERSTANDING_AGENT.value,
|
|
1549
|
+
"title": "Agent contract",
|
|
1550
|
+
"summary": (
|
|
1551
|
+
f"{len(contract.get('tools') or [])} tools, "
|
|
1552
|
+
f"{len(contract.get('hard_constraints') or [])} constraints and "
|
|
1553
|
+
f"{len(contract.get('real_use_cases') or [])} use cases"
|
|
1554
|
+
),
|
|
1555
|
+
"kind": "contract",
|
|
1556
|
+
"data": _presentation_value(contract),
|
|
1557
|
+
"updated_at": datetime.fromtimestamp(
|
|
1558
|
+
(root / "contract.json").stat().st_mtime, timezone.utc
|
|
1559
|
+
).isoformat(),
|
|
1560
|
+
}
|
|
1561
|
+
)
|
|
1562
|
+
environment = _json_artifact(root / "environment.json")
|
|
1563
|
+
if isinstance(environment, dict):
|
|
1564
|
+
visible = _presentation_value(
|
|
1565
|
+
{
|
|
1566
|
+
key: value
|
|
1567
|
+
for key, value in environment.items()
|
|
1568
|
+
if key
|
|
1569
|
+
not in {
|
|
1570
|
+
"source",
|
|
1571
|
+
"compose_file",
|
|
1572
|
+
"compose_override_file",
|
|
1573
|
+
"internal_overrides",
|
|
1574
|
+
"runtime_trace_path",
|
|
1575
|
+
"source_fingerprint",
|
|
1576
|
+
}
|
|
1577
|
+
}
|
|
1578
|
+
)
|
|
1579
|
+
outputs.append(
|
|
1580
|
+
{
|
|
1581
|
+
"id": "environment",
|
|
1582
|
+
"stage": HarnessStage.GENERATING_ENVIRONMENT.value,
|
|
1583
|
+
"title": "Execution environment",
|
|
1584
|
+
"summary": (
|
|
1585
|
+
f"{len(environment.get('services') or [])} services ready"
|
|
1586
|
+
+ (
|
|
1587
|
+
f" in {environment.get('provision_seconds')}s"
|
|
1588
|
+
if environment.get("provision_seconds") is not None
|
|
1589
|
+
else ""
|
|
1590
|
+
)
|
|
1591
|
+
),
|
|
1592
|
+
"kind": "environment",
|
|
1593
|
+
"data": visible,
|
|
1594
|
+
"updated_at": datetime.fromtimestamp(
|
|
1595
|
+
(root / "environment.json").stat().st_mtime, timezone.utc
|
|
1596
|
+
).isoformat(),
|
|
1597
|
+
}
|
|
1598
|
+
)
|
|
1599
|
+
scenarios = _json_artifact(root / "scenarios.json")
|
|
1600
|
+
if isinstance(scenarios, list):
|
|
1601
|
+
outputs.append(
|
|
1602
|
+
{
|
|
1603
|
+
"id": "scenarios",
|
|
1604
|
+
"stage": HarnessStage.GENERATING_SCENARIOS.value,
|
|
1605
|
+
"title": "Generated scenarios",
|
|
1606
|
+
"summary": f"{len(scenarios)} grounded scenarios",
|
|
1607
|
+
"kind": "scenarios",
|
|
1608
|
+
"data": scenarios,
|
|
1609
|
+
"updated_at": datetime.fromtimestamp(
|
|
1610
|
+
(root / "scenarios.json").stat().st_mtime, timezone.utc
|
|
1611
|
+
).isoformat(),
|
|
1612
|
+
}
|
|
1613
|
+
)
|
|
1614
|
+
results = _json_artifact(root / "results.json")
|
|
1615
|
+
if isinstance(results, dict):
|
|
1616
|
+
outputs.append(
|
|
1617
|
+
{
|
|
1618
|
+
"id": "results",
|
|
1619
|
+
"stage": HarnessStage.GRADING.value,
|
|
1620
|
+
"title": "Run results",
|
|
1621
|
+
"summary": "Scenario execution and grading evidence",
|
|
1622
|
+
"kind": "results",
|
|
1623
|
+
"data": results,
|
|
1624
|
+
"updated_at": datetime.fromtimestamp(
|
|
1625
|
+
(root / "results.json").stat().st_mtime, timezone.utc
|
|
1626
|
+
).isoformat(),
|
|
1627
|
+
}
|
|
1628
|
+
)
|
|
1629
|
+
return outputs
|
|
1630
|
+
|
|
1631
|
+
|
|
1632
|
+
def _presentation_value(value: Any, key: str = "") -> Any:
|
|
1633
|
+
lowered = key.lower()
|
|
1634
|
+
if any(marker in lowered for marker in ("password", "secret", "token", "api_key")):
|
|
1635
|
+
return "[redacted]"
|
|
1636
|
+
if isinstance(value, dict):
|
|
1637
|
+
return {
|
|
1638
|
+
str(item_key): _presentation_value(item, str(item_key))
|
|
1639
|
+
for item_key, item in value.items()
|
|
1640
|
+
}
|
|
1641
|
+
if isinstance(value, list):
|
|
1642
|
+
return [_presentation_value(item, key) for item in value]
|
|
1643
|
+
if isinstance(value, str) and "://" in value and "@" in value:
|
|
1644
|
+
scheme, remainder = value.split("://", 1)
|
|
1645
|
+
# Prose can contain an email before a later URL. Only redact when the URI
|
|
1646
|
+
# remainder itself contains userinfo.
|
|
1647
|
+
if "@" in remainder:
|
|
1648
|
+
return f"{scheme}://[redacted]@{remainder.split('@', 1)[1]}"
|
|
1649
|
+
return value
|
|
1650
|
+
|
|
1651
|
+
|
|
1652
|
+
def _read_jsonl(path: Path) -> list[dict[str, Any]]:
|
|
1653
|
+
if not path.is_file():
|
|
1654
|
+
return []
|
|
1655
|
+
records: list[dict[str, Any]] = []
|
|
1656
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
1657
|
+
try:
|
|
1658
|
+
value = json.loads(line)
|
|
1659
|
+
except ValueError:
|
|
1660
|
+
continue
|
|
1661
|
+
if isinstance(value, dict):
|
|
1662
|
+
records.append(value)
|
|
1663
|
+
return records
|
|
1664
|
+
|
|
1665
|
+
|
|
1666
|
+
def _adjustments(directory: Path) -> list[dict[str, Any]]:
|
|
1667
|
+
requested = _read_jsonl(directory / "adjustments.jsonl")
|
|
1668
|
+
acknowledgements = {
|
|
1669
|
+
item.get("adjustment_id"): item
|
|
1670
|
+
for item in _read_jsonl(directory / "adjustment-status.jsonl")
|
|
1671
|
+
}
|
|
1672
|
+
return [
|
|
1673
|
+
{**item, **acknowledgements.get(item.get("adjustment_id"), {})}
|
|
1674
|
+
for item in requested
|
|
1675
|
+
]
|
|
1676
|
+
|
|
1677
|
+
|
|
1678
|
+
def _scenario_delta(instruction: str) -> int | None:
|
|
1679
|
+
match = re.search(
|
|
1680
|
+
r"\b(?:add|create|generate|write)\s+(\d{1,3})\s+(?:more\s+)?scenarios?\b",
|
|
1681
|
+
instruction,
|
|
1682
|
+
re.IGNORECASE,
|
|
1683
|
+
)
|
|
1684
|
+
if not match:
|
|
1685
|
+
return None
|
|
1686
|
+
return min(100, int(match.group(1)))
|
|
1687
|
+
|
|
1688
|
+
|
|
1689
|
+
def _adjustment_stage(instruction: str, current_stage: str) -> str:
|
|
1690
|
+
lowered = instruction.lower()
|
|
1691
|
+
if any(word in lowered for word in ("scenario", "persona", "test case")):
|
|
1692
|
+
return "scenarios"
|
|
1693
|
+
if any(
|
|
1694
|
+
word in lowered
|
|
1695
|
+
for word in ("environment", "database", "service", "seed", "test data")
|
|
1696
|
+
):
|
|
1697
|
+
return "environment"
|
|
1698
|
+
if any(word in lowered for word in ("contract", "tool", "capability")):
|
|
1699
|
+
return "understand"
|
|
1700
|
+
return {
|
|
1701
|
+
HarnessStage.UNDERSTANDING_AGENT.value: "understand",
|
|
1702
|
+
HarnessStage.GENERATING_ENVIRONMENT.value: "environment",
|
|
1703
|
+
HarnessStage.BUILDING_ENVIRONMENT.value: "environment",
|
|
1704
|
+
HarnessStage.VALIDATING_ENVIRONMENT.value: "environment",
|
|
1705
|
+
HarnessStage.GENERATING_SCENARIOS.value: "scenarios",
|
|
1706
|
+
HarnessStage.VALIDATING_SCENARIOS.value: "scenarios",
|
|
1707
|
+
}.get(current_stage, "scenarios")
|
|
1708
|
+
|
|
1709
|
+
|
|
1710
|
+
def _stage_from_events(events: list[dict[str, Any]]) -> str | None:
|
|
1711
|
+
for event in reversed(events):
|
|
1712
|
+
stage = event.get("payload", {}).get("stage")
|
|
1713
|
+
if stage:
|
|
1714
|
+
aliases = {
|
|
1715
|
+
"understand": HarnessStage.UNDERSTANDING_AGENT.value,
|
|
1716
|
+
"environment": HarnessStage.GENERATING_ENVIRONMENT.value,
|
|
1717
|
+
"scenarios": HarnessStage.GENERATING_SCENARIOS.value,
|
|
1718
|
+
"run": HarnessStage.RUNNING.value,
|
|
1719
|
+
"calls": HarnessStage.RUNNING.value,
|
|
1720
|
+
}
|
|
1721
|
+
return aliases.get(
|
|
1722
|
+
stage, stage if stage in HarnessStage._value2member_map_ else None
|
|
1723
|
+
)
|
|
1724
|
+
return None
|
|
1725
|
+
|
|
1726
|
+
|
|
1727
|
+
def _process_identity(pid: int) -> str:
|
|
1728
|
+
"""Return a PID-reuse-safe Linux process identity for orphan ownership."""
|
|
1729
|
+
try:
|
|
1730
|
+
# Field 22 is process start time in clock ticks since boot. ``comm`` may contain spaces,
|
|
1731
|
+
# so split only after the final closing parenthesis instead of indexing the whole line.
|
|
1732
|
+
stat = Path(f"/proc/{pid}/stat").read_text(encoding="utf-8")
|
|
1733
|
+
fields = stat.rsplit(")", 1)[1].strip().split()
|
|
1734
|
+
start_ticks = fields[19]
|
|
1735
|
+
boot_id = (
|
|
1736
|
+
Path("/proc/sys/kernel/random/boot_id").read_text(encoding="utf-8").strip()
|
|
1737
|
+
)
|
|
1738
|
+
return f"{boot_id}:{pid}:{start_ticks}"
|
|
1739
|
+
except (OSError, IndexError, ValueError):
|
|
1740
|
+
return ""
|
|
1741
|
+
|
|
1742
|
+
|
|
1743
|
+
def _read_json(path: Path) -> dict[str, Any]:
|
|
1744
|
+
try:
|
|
1745
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
1746
|
+
except (OSError, ValueError):
|
|
1747
|
+
return {}
|
|
1748
|
+
|
|
1749
|
+
|
|
1750
|
+
def _read_json_value(path: Path) -> Any:
|
|
1751
|
+
try:
|
|
1752
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
1753
|
+
except (OSError, ValueError):
|
|
1754
|
+
return None
|
|
1755
|
+
|
|
1756
|
+
|
|
1757
|
+
def _stage_outputs(run_root: Path) -> list[dict[str, Any]]:
|
|
1758
|
+
"""Return bounded, secret-safe snapshots for the platform's live run UI."""
|
|
1759
|
+
outputs: list[dict[str, Any]] = []
|
|
1760
|
+
contract = _read_json_value(run_root / "contract.json")
|
|
1761
|
+
if isinstance(contract, dict):
|
|
1762
|
+
tools = [
|
|
1763
|
+
{"name": str(tool.get("name") or "")}
|
|
1764
|
+
for tool in contract.get("tools", [])
|
|
1765
|
+
if isinstance(tool, dict) and tool.get("name")
|
|
1766
|
+
]
|
|
1767
|
+
data = {
|
|
1768
|
+
"one_liner": str(contract.get("one_liner") or ""),
|
|
1769
|
+
"modality": str(contract.get("modality") or "unknown"),
|
|
1770
|
+
"runtime": contract.get("runtime") or {},
|
|
1771
|
+
"tools": tools,
|
|
1772
|
+
"hard_constraints": [
|
|
1773
|
+
str(item) for item in contract.get("hard_constraints", [])
|
|
1774
|
+
],
|
|
1775
|
+
}
|
|
1776
|
+
outputs.append(
|
|
1777
|
+
{
|
|
1778
|
+
"id": "contract",
|
|
1779
|
+
"kind": "contract",
|
|
1780
|
+
"title": "Agent contract",
|
|
1781
|
+
"summary": f"{len(tools)} tools · {data['modality']} modality",
|
|
1782
|
+
"data": data,
|
|
1783
|
+
}
|
|
1784
|
+
)
|
|
1785
|
+
|
|
1786
|
+
environment = _read_json_value(run_root / "environment.json")
|
|
1787
|
+
if isinstance(environment, dict):
|
|
1788
|
+
services = [str(item) for item in environment.get("services", [])]
|
|
1789
|
+
# Endpoint values are useful operational feedback, but credentials embedded in a URI
|
|
1790
|
+
# must never reach the platform response.
|
|
1791
|
+
overrides = {
|
|
1792
|
+
str(name): re.sub(r"(://)[^/@]+@", r"\1***@", str(value))
|
|
1793
|
+
for name, value in dict(environment.get("overrides") or {}).items()
|
|
1794
|
+
}
|
|
1795
|
+
outputs.append(
|
|
1796
|
+
{
|
|
1797
|
+
"id": "environment",
|
|
1798
|
+
"kind": "environment",
|
|
1799
|
+
"title": "Execution environment",
|
|
1800
|
+
"summary": f"{len(services)} services ready",
|
|
1801
|
+
"data": {
|
|
1802
|
+
"services": services,
|
|
1803
|
+
"project": str(environment.get("project") or ""),
|
|
1804
|
+
"managed": bool(environment.get("managed")),
|
|
1805
|
+
"overrides": overrides,
|
|
1806
|
+
},
|
|
1807
|
+
}
|
|
1808
|
+
)
|
|
1809
|
+
|
|
1810
|
+
scenarios = _read_json_value(run_root / "scenarios.json")
|
|
1811
|
+
if isinstance(scenarios, list):
|
|
1812
|
+
data = [
|
|
1813
|
+
{
|
|
1814
|
+
"name": str(item.get("name") or "scenario"),
|
|
1815
|
+
"instruction": str(item.get("instruction") or ""),
|
|
1816
|
+
"use_case": str(item.get("use_case") or ""),
|
|
1817
|
+
}
|
|
1818
|
+
for item in scenarios
|
|
1819
|
+
if isinstance(item, dict)
|
|
1820
|
+
]
|
|
1821
|
+
outputs.append(
|
|
1822
|
+
{
|
|
1823
|
+
"id": "scenarios",
|
|
1824
|
+
"kind": "scenarios",
|
|
1825
|
+
"title": "Generated scenarios",
|
|
1826
|
+
"summary": f"{len(data)} grounded scenarios",
|
|
1827
|
+
"data": data,
|
|
1828
|
+
}
|
|
1829
|
+
)
|
|
1830
|
+
|
|
1831
|
+
platform = _read_json_value(run_root / "platform.json")
|
|
1832
|
+
if isinstance(platform, dict):
|
|
1833
|
+
run_test_id = str(platform.get("run_test_id") or "").strip()
|
|
1834
|
+
execution_id = str(platform.get("test_execution_id") or "").strip()
|
|
1835
|
+
if run_test_id:
|
|
1836
|
+
url = (
|
|
1837
|
+
f"/dashboard/simulate/test/{run_test_id}/{execution_id}"
|
|
1838
|
+
if execution_id
|
|
1839
|
+
else f"/dashboard/simulate/test/{run_test_id}/runs"
|
|
1840
|
+
)
|
|
1841
|
+
outputs.append(
|
|
1842
|
+
{
|
|
1843
|
+
"id": "simulation",
|
|
1844
|
+
"kind": "simulation",
|
|
1845
|
+
"title": "Simulation results",
|
|
1846
|
+
"summary": "Open this run in the simulation view",
|
|
1847
|
+
"data": {
|
|
1848
|
+
"url": url,
|
|
1849
|
+
"run_test_id": run_test_id,
|
|
1850
|
+
"test_execution_id": execution_id,
|
|
1851
|
+
},
|
|
1852
|
+
}
|
|
1853
|
+
)
|
|
1854
|
+
return outputs
|
|
1855
|
+
|
|
1856
|
+
|
|
1857
|
+
def _write_json(path: Path, value: dict[str, Any]) -> None:
|
|
1858
|
+
temporary = path.with_suffix(".tmp")
|
|
1859
|
+
temporary.write_text(
|
|
1860
|
+
json.dumps(value, indent=2, default=str) + "\n", encoding="utf-8"
|
|
1861
|
+
)
|
|
1862
|
+
temporary.replace(path)
|
|
1863
|
+
|
|
1864
|
+
|
|
1865
|
+
def _now() -> str:
|
|
1866
|
+
return datetime.now(timezone.utc).isoformat()
|
|
1867
|
+
|
|
1868
|
+
|
|
1869
|
+
_TERMINAL_STAGES = {
|
|
1870
|
+
HarnessStage.COMPLETED.value,
|
|
1871
|
+
HarnessStage.FAILED.value,
|
|
1872
|
+
HarnessStage.CANCELED.value,
|
|
1873
|
+
}
|
|
1874
|
+
|
|
1875
|
+
|
|
1876
|
+
def _authorize(authorization: str | None = Header(default=None)) -> None:
|
|
1877
|
+
expected = os.getenv("ALK_SANDBOX_TOKEN")
|
|
1878
|
+
if expected and authorization != f"Bearer {expected}":
|
|
1879
|
+
raise HTTPException(
|
|
1880
|
+
status_code=status.HTTP_401_UNAUTHORIZED, detail="invalid token"
|
|
1881
|
+
)
|
|
1882
|
+
|
|
1883
|
+
|
|
1884
|
+
def create_app(root: Path | None = None) -> FastAPI:
|
|
1885
|
+
app = FastAPI(title="ALK local sandbox", version="1")
|
|
1886
|
+
sandbox = LocalSandbox(
|
|
1887
|
+
root or Path(os.getenv("ALK_SANDBOX_ROOT", "./artifacts/sandbox")),
|
|
1888
|
+
max_concurrency=int(os.getenv("ALK_SANDBOX_MAX_CONCURRENCY", "2")),
|
|
1889
|
+
)
|
|
1890
|
+
|
|
1891
|
+
@app.get("/health")
|
|
1892
|
+
async def health() -> dict[str, str]:
|
|
1893
|
+
return {"status": "ok", "provider": "local-process"}
|
|
1894
|
+
|
|
1895
|
+
@app.post(
|
|
1896
|
+
"/v1/sources",
|
|
1897
|
+
response_model=UploadedSourceResponse,
|
|
1898
|
+
dependencies=[Depends(_authorize)],
|
|
1899
|
+
status_code=status.HTTP_201_CREATED,
|
|
1900
|
+
)
|
|
1901
|
+
async def upload_source(
|
|
1902
|
+
files: list[UploadFile] = File(...),
|
|
1903
|
+
paths: list[str] = Form(...),
|
|
1904
|
+
name: str = Form(default="uploaded-agent"),
|
|
1905
|
+
) -> UploadedSourceResponse:
|
|
1906
|
+
return await sandbox.upload_source(files, paths, name)
|
|
1907
|
+
|
|
1908
|
+
@app.post(
|
|
1909
|
+
"/v1/secret-files",
|
|
1910
|
+
response_model=UploadedSecretFileResponse,
|
|
1911
|
+
dependencies=[Depends(_authorize)],
|
|
1912
|
+
status_code=status.HTTP_201_CREATED,
|
|
1913
|
+
)
|
|
1914
|
+
async def upload_secret_file(
|
|
1915
|
+
file: UploadFile = File(...),
|
|
1916
|
+
environment_name: str = Form(...),
|
|
1917
|
+
) -> UploadedSecretFileResponse:
|
|
1918
|
+
return await sandbox.upload_secret_file(file, environment_name)
|
|
1919
|
+
|
|
1920
|
+
@app.post(
|
|
1921
|
+
"/v1/preflight",
|
|
1922
|
+
response_model=SandboxPreflightResponse,
|
|
1923
|
+
dependencies=[Depends(_authorize)],
|
|
1924
|
+
)
|
|
1925
|
+
async def preflight(
|
|
1926
|
+
request: SandboxPreflightRequest,
|
|
1927
|
+
) -> SandboxPreflightResponse:
|
|
1928
|
+
return sandbox.preflight(request)
|
|
1929
|
+
|
|
1930
|
+
@app.post(
|
|
1931
|
+
"/v1/jobs",
|
|
1932
|
+
response_model=SandboxJobResponse,
|
|
1933
|
+
dependencies=[Depends(_authorize)],
|
|
1934
|
+
)
|
|
1935
|
+
async def submit(request: LocalSandboxRequest) -> SandboxJobResponse:
|
|
1936
|
+
return sandbox.submit(request)
|
|
1937
|
+
|
|
1938
|
+
@app.get(
|
|
1939
|
+
"/v1/jobs",
|
|
1940
|
+
response_model=list[SandboxJobResponse],
|
|
1941
|
+
dependencies=[Depends(_authorize)],
|
|
1942
|
+
)
|
|
1943
|
+
async def list_jobs() -> list[SandboxJobResponse]:
|
|
1944
|
+
return sandbox.list()
|
|
1945
|
+
|
|
1946
|
+
@app.get(
|
|
1947
|
+
"/v1/jobs/{job_id}",
|
|
1948
|
+
response_model=SandboxJobResponse,
|
|
1949
|
+
dependencies=[Depends(_authorize)],
|
|
1950
|
+
)
|
|
1951
|
+
async def get_job(job_id: str) -> SandboxJobResponse:
|
|
1952
|
+
try:
|
|
1953
|
+
return sandbox.get(job_id)
|
|
1954
|
+
except KeyError as exc:
|
|
1955
|
+
raise HTTPException(status_code=404, detail="job not found") from exc
|
|
1956
|
+
|
|
1957
|
+
@app.post(
|
|
1958
|
+
"/v1/jobs/{job_id}/rerun",
|
|
1959
|
+
response_model=SandboxJobResponse,
|
|
1960
|
+
dependencies=[Depends(_authorize)],
|
|
1961
|
+
)
|
|
1962
|
+
async def rerun_job(
|
|
1963
|
+
job_id: str, request: SandboxRerunRequest
|
|
1964
|
+
) -> SandboxJobResponse:
|
|
1965
|
+
try:
|
|
1966
|
+
return sandbox.rerun(job_id, request)
|
|
1967
|
+
except KeyError as exc:
|
|
1968
|
+
raise HTTPException(status_code=404, detail="job not found") from exc
|
|
1969
|
+
|
|
1970
|
+
@app.post(
|
|
1971
|
+
"/v1/jobs/{job_id}/cancel",
|
|
1972
|
+
response_model=SandboxJobResponse,
|
|
1973
|
+
dependencies=[Depends(_authorize)],
|
|
1974
|
+
)
|
|
1975
|
+
async def cancel_job(job_id: str) -> SandboxJobResponse:
|
|
1976
|
+
try:
|
|
1977
|
+
return await sandbox.cancel(job_id)
|
|
1978
|
+
except KeyError as exc:
|
|
1979
|
+
raise HTTPException(status_code=404, detail="job not found") from exc
|
|
1980
|
+
|
|
1981
|
+
@app.post(
|
|
1982
|
+
"/v1/jobs/{job_id}/adjust",
|
|
1983
|
+
response_model=SandboxJobResponse,
|
|
1984
|
+
dependencies=[Depends(_authorize)],
|
|
1985
|
+
)
|
|
1986
|
+
async def adjust_job(
|
|
1987
|
+
job_id: str, request: SandboxAdjustmentRequest
|
|
1988
|
+
) -> SandboxJobResponse:
|
|
1989
|
+
try:
|
|
1990
|
+
return sandbox.adjust(job_id, request)
|
|
1991
|
+
except KeyError as exc:
|
|
1992
|
+
raise HTTPException(status_code=404, detail="job not found") from exc
|
|
1993
|
+
|
|
1994
|
+
return app
|
|
1995
|
+
|
|
1996
|
+
|
|
1997
|
+
app = create_app()
|
|
1998
|
+
|
|
1999
|
+
|
|
2000
|
+
def main() -> None:
|
|
2001
|
+
import uvicorn
|
|
2002
|
+
|
|
2003
|
+
uvicorn.run(
|
|
2004
|
+
"fi.alk.harness.sandbox_server:app",
|
|
2005
|
+
host=os.getenv("ALK_SANDBOX_HOST", "127.0.0.1"),
|
|
2006
|
+
port=int(os.getenv("ALK_SANDBOX_PORT", "8788")),
|
|
2007
|
+
)
|
|
2008
|
+
|
|
2009
|
+
|
|
2010
|
+
if __name__ == "__main__":
|
|
2011
|
+
main()
|