agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,692 @@
|
|
|
1
|
+
"""Reporting a run to the platform, so it appears where every other run does.
|
|
2
|
+
|
|
3
|
+
The platform already has somewhere to put this. Its simulate pages read `RunTest`,
|
|
4
|
+
`TestExecution` and `CallExecution`, and the ingestion API that the hosted runner posts to builds
|
|
5
|
+
exactly those. So a harness run is not shown by drawing it again somewhere else; it is shown by
|
|
6
|
+
walking the same API, and the pages that already exist render it unchanged.
|
|
7
|
+
|
|
8
|
+
provision ──► a RunTest for this session, once
|
|
9
|
+
start ──► a TestExecution, once per run, so running twice gives two runs
|
|
10
|
+
batch ──► a CallExecution per scenario
|
|
11
|
+
result ──► what the scenario did, one call at a time
|
|
12
|
+
recording ──► the audio, where a spoken run left any
|
|
13
|
+
|
|
14
|
+
What is deliberately *not* sent: interruption counts, talk ratio, latency, scores. The backend
|
|
15
|
+
derives those from the transcript it is given, and a second implementation here would drift from
|
|
16
|
+
the one the rest of the platform is measured by. This reports only what the run observed.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import hashlib
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import re
|
|
25
|
+
import urllib.error
|
|
26
|
+
import urllib.parse
|
|
27
|
+
import urllib.request
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
from datetime import datetime, timezone
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
# Where runs are reported. Its own variables, because reporting and evaluating go to different
|
|
34
|
+
# places: the eval templates a run is scored against live on the hosted platform, while the runs
|
|
35
|
+
# themselves belong wherever the person is looking at them -- usually the backend beside this
|
|
36
|
+
# harness. Sharing FI_* for both means one of the two is always pointed at the wrong host.
|
|
37
|
+
# FI_* is the fallback, so a setup that genuinely uses one platform for both still works unchanged.
|
|
38
|
+
BASE_URL = ("HARNESS_PLATFORM_URL", "FI_BASE_URL")
|
|
39
|
+
API_KEY = ("HARNESS_PLATFORM_API_KEY", "FI_API_KEY")
|
|
40
|
+
SECRET_KEY = ("HARNESS_PLATFORM_SECRET_KEY", "FI_SECRET_KEY")
|
|
41
|
+
WORKSPACE_ID = "HARNESS_PLATFORM_WORKSPACE_ID"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _setting(names: tuple[str, ...]) -> str:
|
|
45
|
+
"""The first of these that is set, so the specific name wins over the shared one."""
|
|
46
|
+
for name in names:
|
|
47
|
+
found = os.environ.get(name, "").strip()
|
|
48
|
+
if found:
|
|
49
|
+
return found
|
|
50
|
+
return ""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
INGESTION = "/simulate/api/alk-simulate"
|
|
54
|
+
|
|
55
|
+
# Django appends a slash and cannot redirect a POST while keeping its body, so every path here
|
|
56
|
+
# carries one already. Without it the call fails as a 500 that reads like a server fault.
|
|
57
|
+
TIMEOUT_SECONDS = 120.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _open(request: urllib.request.Request, timeout: float = TIMEOUT_SECONDS):
|
|
61
|
+
"""Open an ingestion request without sending loopback traffic to a proxy.
|
|
62
|
+
|
|
63
|
+
Developer machines commonly export an HTTP(S) proxy for model/provider
|
|
64
|
+
traffic. urllib applies it to the local platform too unless NO_PROXY happens
|
|
65
|
+
to be configured, producing an unrelated proxy 400 with an empty body.
|
|
66
|
+
Remote platform URLs retain normal proxy behavior.
|
|
67
|
+
"""
|
|
68
|
+
host = (urllib.parse.urlsplit(request.full_url).hostname or "").lower()
|
|
69
|
+
if host in {"127.0.0.1", "localhost", "::1"}:
|
|
70
|
+
return urllib.request.build_opener(urllib.request.ProxyHandler({})).open(
|
|
71
|
+
request, timeout=timeout
|
|
72
|
+
)
|
|
73
|
+
return urllib.request.urlopen(request, timeout=timeout)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class PlatformError(RuntimeError):
|
|
77
|
+
"""The platform refused or could not be reached, with enough detail to act on."""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class Reported:
|
|
82
|
+
"""Where a run ended up, so a caller can link to it."""
|
|
83
|
+
|
|
84
|
+
run_test_id: str = ""
|
|
85
|
+
test_execution_id: str = ""
|
|
86
|
+
calls: dict[str, str] = field(default_factory=dict)
|
|
87
|
+
problems: list[str] = field(default_factory=list)
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def url(self) -> str:
|
|
91
|
+
"""Where this exact execution is on the platform."""
|
|
92
|
+
if not self.run_test_id:
|
|
93
|
+
return ""
|
|
94
|
+
if self.test_execution_id:
|
|
95
|
+
return (
|
|
96
|
+
f"/dashboard/simulate/test/{self.run_test_id}/{self.test_execution_id}"
|
|
97
|
+
)
|
|
98
|
+
return f"/dashboard/simulate/test/{self.run_test_id}/runs"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def display_run_name(agent: str, *, now: datetime | None = None) -> str:
|
|
102
|
+
"""A readable, unique platform name without exposing a harness UUID."""
|
|
103
|
+
words = re.sub(r"[-_]+", " ", str(agent or "agent")).strip()
|
|
104
|
+
title = " ".join(
|
|
105
|
+
word if word.isupper() else word.capitalize() for word in words.split()
|
|
106
|
+
)
|
|
107
|
+
timestamp = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
|
|
108
|
+
return f"{title or 'Agent'} · {timestamp:%d %b %Y %H:%M UTC}"[:255]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def display_scenario_name(scenario: Any) -> str:
|
|
112
|
+
"""Prefer the scenario's behavior over an internal slug or repeated run prefix."""
|
|
113
|
+
for value in (
|
|
114
|
+
getattr(scenario, "use_case", ""),
|
|
115
|
+
getattr(scenario, "tests", ""),
|
|
116
|
+
getattr(scenario, "name", ""),
|
|
117
|
+
):
|
|
118
|
+
text = str(value or "").strip().rstrip(".")
|
|
119
|
+
if text:
|
|
120
|
+
if value == getattr(scenario, "name", ""):
|
|
121
|
+
text = re.sub(r"[-_]+", " ", text)
|
|
122
|
+
text = text[:1].upper() + text[1:]
|
|
123
|
+
return text[:255]
|
|
124
|
+
return "Scenario"
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def configured() -> str:
|
|
128
|
+
"""Why a run cannot be reported, or an empty string when it can."""
|
|
129
|
+
missing = [
|
|
130
|
+
names[0] for names in (BASE_URL, API_KEY, SECRET_KEY) if not _setting(names)
|
|
131
|
+
]
|
|
132
|
+
if missing:
|
|
133
|
+
return f"{', '.join(missing)} not set, so this run stays local"
|
|
134
|
+
return ""
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class Platform:
|
|
138
|
+
"""The ingestion API, as the few calls a run actually makes."""
|
|
139
|
+
|
|
140
|
+
def __init__(self, base: str = "", key: str = "", secret: str = "") -> None:
|
|
141
|
+
self.base = (base or _setting(BASE_URL)).rstrip("/")
|
|
142
|
+
self.key = key or _setting(API_KEY)
|
|
143
|
+
self.secret = secret or _setting(SECRET_KEY)
|
|
144
|
+
|
|
145
|
+
def _headers(self) -> dict[str, str]:
|
|
146
|
+
headers = {
|
|
147
|
+
"X-Api-Key": self.key,
|
|
148
|
+
"X-Secret-Key": self.secret,
|
|
149
|
+
}
|
|
150
|
+
workspace_id = os.environ.get(WORKSPACE_ID, "").strip()
|
|
151
|
+
if workspace_id:
|
|
152
|
+
headers["X-Workspace-Id"] = workspace_id
|
|
153
|
+
return headers
|
|
154
|
+
|
|
155
|
+
def _call(
|
|
156
|
+
self, path: str, payload: dict[str, Any], method: str = "POST"
|
|
157
|
+
) -> dict[str, Any]:
|
|
158
|
+
request = urllib.request.Request(
|
|
159
|
+
f"{self.base}{INGESTION}{path}",
|
|
160
|
+
data=json.dumps(payload).encode(),
|
|
161
|
+
headers={"Content-Type": "application/json", **self._headers()},
|
|
162
|
+
method=method,
|
|
163
|
+
)
|
|
164
|
+
try:
|
|
165
|
+
with _open(request) as answer:
|
|
166
|
+
body = json.loads(answer.read().decode() or "{}")
|
|
167
|
+
except urllib.error.HTTPError as refused:
|
|
168
|
+
detail = refused.read().decode(errors="replace")[:400]
|
|
169
|
+
raise PlatformError(
|
|
170
|
+
f"{method} {path} failed ({refused.code}): {detail}"
|
|
171
|
+
) from refused
|
|
172
|
+
except Exception as unreachable: # noqa: BLE001 - reported, not handled
|
|
173
|
+
raise PlatformError(
|
|
174
|
+
f"{method} {path} could not be sent: {unreachable}"
|
|
175
|
+
) from unreachable
|
|
176
|
+
# The platform wraps every answer; unwrap it here so callers read the payload itself.
|
|
177
|
+
return body.get("result", body) if isinstance(body, dict) else {}
|
|
178
|
+
|
|
179
|
+
def provision(
|
|
180
|
+
self, name: str, personas: list[dict[str, Any]], modality: str = "text"
|
|
181
|
+
) -> dict[str, Any]:
|
|
182
|
+
agent_name = name.split(" · ", 1)[0].strip() or "ALK agent"
|
|
183
|
+
return self._call(
|
|
184
|
+
"/run-tests/provision/",
|
|
185
|
+
{
|
|
186
|
+
"name": name,
|
|
187
|
+
"agent_name": agent_name,
|
|
188
|
+
"personas": personas,
|
|
189
|
+
"modality": modality,
|
|
190
|
+
},
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
def start(
|
|
194
|
+
self,
|
|
195
|
+
run_test_id: str,
|
|
196
|
+
scenario_ids: list[str] | None = None,
|
|
197
|
+
*,
|
|
198
|
+
harness_job_id: str = "",
|
|
199
|
+
scenario_selectors: list[dict[str, str]] | None = None,
|
|
200
|
+
) -> dict[str, Any]:
|
|
201
|
+
payload = {"scenario_ids": scenario_ids} if scenario_ids else {}
|
|
202
|
+
if harness_job_id:
|
|
203
|
+
payload["harness_job_id"] = harness_job_id
|
|
204
|
+
if scenario_selectors:
|
|
205
|
+
payload["scenario_selectors"] = scenario_selectors
|
|
206
|
+
return self._call(f"/run-tests/{run_test_id}/test-executions/", payload)
|
|
207
|
+
|
|
208
|
+
def batch(self, test_execution_id: str, count: int) -> dict[str, Any]:
|
|
209
|
+
return self._call(
|
|
210
|
+
f"/test-executions/{test_execution_id}/batch/", {"count": count}
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
def result(self, call_execution_id: str, payload: dict[str, Any]) -> dict[str, Any]:
|
|
214
|
+
return self._call(
|
|
215
|
+
f"/call-executions/{call_execution_id}/result/", payload, method="PATCH"
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
def ongoing(self, call_execution_id: str) -> dict[str, Any]:
|
|
219
|
+
"""Mark one pre-allocated call as started using the established ingestion route."""
|
|
220
|
+
return self._call(
|
|
221
|
+
f"/call-executions/{call_execution_id}/status/",
|
|
222
|
+
{"status": "ongoing"},
|
|
223
|
+
method="PATCH",
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
def recording(self, call_execution_id: str, audio: Path) -> dict[str, Any]:
|
|
227
|
+
"""Send one call's audio, as the multipart upload the endpoint expects.
|
|
228
|
+
|
|
229
|
+
Built by hand rather than with a library: this is the only multipart request the harness
|
|
230
|
+
makes, and a dependency for one boundary string is not worth carrying.
|
|
231
|
+
"""
|
|
232
|
+
edge = "----harness" + os.urandom(8).hex()
|
|
233
|
+
content = audio.read_bytes()
|
|
234
|
+
digest = hashlib.sha256(content).hexdigest()
|
|
235
|
+
field = (
|
|
236
|
+
f"--{edge}\r\n"
|
|
237
|
+
f'Content-Disposition: form-data; name="file"; filename="{audio.name}"\r\n'
|
|
238
|
+
"Content-Type: audio/wav\r\n\r\n"
|
|
239
|
+
).encode()
|
|
240
|
+
checksum = (
|
|
241
|
+
f"\r\n--{edge}\r\n"
|
|
242
|
+
'Content-Disposition: form-data; name="sha256"\r\n\r\n'
|
|
243
|
+
f"{digest}"
|
|
244
|
+
).encode()
|
|
245
|
+
tail = f"\r\n--{edge}--\r\n".encode()
|
|
246
|
+
body = field + content + checksum + tail
|
|
247
|
+
request = urllib.request.Request(
|
|
248
|
+
f"{self.base}{INGESTION}/call-executions/{call_execution_id}/recording/",
|
|
249
|
+
data=body,
|
|
250
|
+
headers={
|
|
251
|
+
"Content-Type": f"multipart/form-data; boundary={edge}",
|
|
252
|
+
**self._headers(),
|
|
253
|
+
},
|
|
254
|
+
)
|
|
255
|
+
try:
|
|
256
|
+
with _open(request) as answer:
|
|
257
|
+
return json.loads(answer.read().decode() or "{}")
|
|
258
|
+
except urllib.error.HTTPError as refused:
|
|
259
|
+
detail = refused.read().decode(errors="replace")[:300]
|
|
260
|
+
raise PlatformError(
|
|
261
|
+
f"recording upload failed ({refused.code}): {detail}"
|
|
262
|
+
) from refused
|
|
263
|
+
except Exception as unreachable: # noqa: BLE001 - reported, not handled
|
|
264
|
+
raise PlatformError(
|
|
265
|
+
f"recording could not be sent: {unreachable}"
|
|
266
|
+
) from unreachable
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def persona_of(scenario: Any) -> dict[str, Any]:
|
|
270
|
+
"""One scenario as the platform's persona record.
|
|
271
|
+
|
|
272
|
+
``persona`` is carried whole so the simulator prompt's placeholder resolves against the same
|
|
273
|
+
person the scenario was written for, rather than a name reconstructed from it.
|
|
274
|
+
"""
|
|
275
|
+
persona = getattr(scenario, "persona", None) or {}
|
|
276
|
+
if hasattr(persona, "model_dump"):
|
|
277
|
+
persona = persona.model_dump()
|
|
278
|
+
elif not isinstance(persona, dict):
|
|
279
|
+
persona = {}
|
|
280
|
+
return {
|
|
281
|
+
"name": str(persona.get("name") or getattr(scenario, "name", "") or "caller")[
|
|
282
|
+
:255
|
|
283
|
+
],
|
|
284
|
+
# The scenario's own key, not its folder name: the key is ASCII-sanitised and falls back
|
|
285
|
+
# to a digest, which the name does not, and this value travels as an HTTP header.
|
|
286
|
+
"scenario_key": str(
|
|
287
|
+
getattr(scenario, "scenario_key", "") or getattr(scenario, "name", "") or ""
|
|
288
|
+
)[:255],
|
|
289
|
+
"scenario_name": display_scenario_name(scenario),
|
|
290
|
+
"role": str(persona.get("role") or persona.get("occupation") or "")[:255],
|
|
291
|
+
"situation": str(getattr(scenario, "instruction", "") or ""),
|
|
292
|
+
"outcome": str(getattr(scenario, "tests", "") or ""),
|
|
293
|
+
"persona": persona,
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
# What the harness calls a speaker, and what a transcript row is called on the platform. Anything
|
|
298
|
+
# unrecognised is the person, because the agent's turns are the ones we name.
|
|
299
|
+
SPEAKERS = {
|
|
300
|
+
"agent": "assistant",
|
|
301
|
+
"assistant": "assistant",
|
|
302
|
+
"bot": "assistant",
|
|
303
|
+
"system": "system",
|
|
304
|
+
"customer": "user",
|
|
305
|
+
"caller": "user",
|
|
306
|
+
"user": "user",
|
|
307
|
+
"tester": "user",
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def segments_of(result: Any) -> list[dict[str, Any]]:
|
|
312
|
+
"""A run's conversation as transcript rows, tool calls included.
|
|
313
|
+
|
|
314
|
+
No timings are invented. A typed run has none to give, and a made-up millisecond would be
|
|
315
|
+
indistinguishable from a measured one to everything downstream that averages them. A spoken
|
|
316
|
+
turn the runner timed carries those times through, because the platform derives duration,
|
|
317
|
+
silence, talk ratio and latency from them and can derive none of it from zeros.
|
|
318
|
+
"""
|
|
319
|
+
rows: list[dict[str, Any]] = []
|
|
320
|
+
for turn in getattr(result, "exchanges", None) or []:
|
|
321
|
+
said = str(turn.get("text") or "").strip()
|
|
322
|
+
if not said:
|
|
323
|
+
continue
|
|
324
|
+
row = {
|
|
325
|
+
"speaker_role": SPEAKERS.get(str(turn.get("speaker", "")).lower(), "user"),
|
|
326
|
+
"content": said,
|
|
327
|
+
}
|
|
328
|
+
for when in ("start_time_ms", "end_time_ms"):
|
|
329
|
+
if turn.get(when) is not None:
|
|
330
|
+
row[when] = int(turn[when])
|
|
331
|
+
rows.append(row)
|
|
332
|
+
for call in getattr(result, "calls_detail", None) or []:
|
|
333
|
+
rows.append(
|
|
334
|
+
{
|
|
335
|
+
"speaker_role": "tool_calls",
|
|
336
|
+
"content": f"{call.get('name', '')}({json.dumps(call.get('arguments', {}), default=str)})",
|
|
337
|
+
}
|
|
338
|
+
)
|
|
339
|
+
outcome = call.get("error") or call.get("result") or ""
|
|
340
|
+
rows.append(
|
|
341
|
+
{
|
|
342
|
+
"speaker_role": "tool_call_result",
|
|
343
|
+
"content": ("refused: " if call.get("refused") else "")
|
|
344
|
+
+ str(outcome)[:4000],
|
|
345
|
+
}
|
|
346
|
+
)
|
|
347
|
+
return rows
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def evaluations_of(result: Any) -> list[dict[str, Any]]:
|
|
351
|
+
"""Everything this run judged, as one list the platform can render per call.
|
|
352
|
+
|
|
353
|
+
Two kinds arrive from different places and mean different things, so both are named and
|
|
354
|
+
kept apart rather than averaged into a verdict. A sub-goal is deterministic: the world was
|
|
355
|
+
left in a state, or it was not. A metric is scored: the run placed it somewhere between
|
|
356
|
+
nothing and everything. Reporting only the first is what made a scored run look unjudged.
|
|
357
|
+
"""
|
|
358
|
+
judged: list[dict[str, Any]] = []
|
|
359
|
+
for check in getattr(result, "checkpoints", None) or []:
|
|
360
|
+
decided_by = str(getattr(check, "by", "") or "")
|
|
361
|
+
# Platform-backed judgements carry ``<template name> (<model>)`` in
|
|
362
|
+
# ``by``. Keep the exact template name on the wire so ingestion can
|
|
363
|
+
# attach the already-computed result to that template/config instead of
|
|
364
|
+
# merely leaving a second, disconnected EvalTemplate in the library.
|
|
365
|
+
platform_template = decided_by.rsplit(" (", 1)[0] if decided_by else ""
|
|
366
|
+
evaluation = {
|
|
367
|
+
"name": getattr(check, "name", ""),
|
|
368
|
+
"kind": getattr(check, "kind", "") or "checkpoint",
|
|
369
|
+
"passed": bool(getattr(check, "passed", False)),
|
|
370
|
+
"reason": str(getattr(check, "detail", ""))[:2000],
|
|
371
|
+
"decided_by": decided_by[:2000],
|
|
372
|
+
"platform_template": platform_template[:2000],
|
|
373
|
+
}
|
|
374
|
+
if getattr(check, "grading_error", False):
|
|
375
|
+
evaluation["grading_error"] = True
|
|
376
|
+
judged.append(evaluation)
|
|
377
|
+
for metric in (getattr(result, "measured", None) or {}).get("metrics") or []:
|
|
378
|
+
if not metric.get("applicable", True):
|
|
379
|
+
continue
|
|
380
|
+
judged.append(
|
|
381
|
+
{
|
|
382
|
+
"name": str(metric.get("name", "")),
|
|
383
|
+
"kind": "metric",
|
|
384
|
+
"score": float(metric.get("score", 0.0) or 0.0),
|
|
385
|
+
"reason": str(metric.get("reason", ""))[:2000],
|
|
386
|
+
}
|
|
387
|
+
)
|
|
388
|
+
return [one for one in judged if one["name"]]
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def conversation_seconds(result: Any) -> int:
|
|
392
|
+
"""Return user-visible conversation time, excluding setup and retry waits."""
|
|
393
|
+
exchanges = getattr(result, "exchanges", None) or []
|
|
394
|
+
starts = [
|
|
395
|
+
float(exchange["start_time_ms"])
|
|
396
|
+
for exchange in exchanges
|
|
397
|
+
if isinstance(exchange, dict)
|
|
398
|
+
and isinstance(exchange.get("start_time_ms"), (int, float))
|
|
399
|
+
]
|
|
400
|
+
ends = [
|
|
401
|
+
float(exchange["end_time_ms"])
|
|
402
|
+
for exchange in exchanges
|
|
403
|
+
if isinstance(exchange, dict)
|
|
404
|
+
and isinstance(exchange.get("end_time_ms"), (int, float))
|
|
405
|
+
]
|
|
406
|
+
if starts and ends:
|
|
407
|
+
return max(0, int((max(ends) - min(starts)) / 1000))
|
|
408
|
+
if not exchanges and getattr(result, "problems", None):
|
|
409
|
+
return 0
|
|
410
|
+
return max(0, int(getattr(result, "seconds", 0) or 0))
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def result_of(result: Any) -> dict[str, Any]:
|
|
414
|
+
"""One scenario's outcome, in the shape the ingestion API takes.
|
|
415
|
+
|
|
416
|
+
Sub-goals travel in ``call_metadata`` rather than as free text: they are what this run
|
|
417
|
+
actually decided, and a page showing one goal per column needs them named and separate.
|
|
418
|
+
"""
|
|
419
|
+
checkpoints = [
|
|
420
|
+
{
|
|
421
|
+
"name": getattr(check, "name", ""),
|
|
422
|
+
"kind": getattr(check, "kind", ""),
|
|
423
|
+
"passed": bool(getattr(check, "passed", False)),
|
|
424
|
+
"detail": str(getattr(check, "detail", ""))[:2000],
|
|
425
|
+
}
|
|
426
|
+
for check in getattr(result, "checkpoints", None) or []
|
|
427
|
+
]
|
|
428
|
+
problems = list(getattr(result, "problems", None) or [])
|
|
429
|
+
grading_failures = list(getattr(result, "grading_failures", None) or [])
|
|
430
|
+
evaluations = evaluations_of(result)
|
|
431
|
+
evaluation_coverage = {
|
|
432
|
+
"expected": len(evaluations),
|
|
433
|
+
"executed": sum(
|
|
434
|
+
1 for evaluation in evaluations if not evaluation.get("grading_error")
|
|
435
|
+
),
|
|
436
|
+
"failed": sum(
|
|
437
|
+
1 for evaluation in evaluations if evaluation.get("grading_error")
|
|
438
|
+
),
|
|
439
|
+
"complete": not grading_failures,
|
|
440
|
+
}
|
|
441
|
+
payload: dict[str, Any] = {
|
|
442
|
+
# A scenario that never ran is not a scenario the agent failed, and the two must not
|
|
443
|
+
# arrive as the same status.
|
|
444
|
+
"status": "failed" if problems else "completed",
|
|
445
|
+
"duration_seconds": conversation_seconds(result),
|
|
446
|
+
"ended_reason": (getattr(result, "ended", "") or "")[:10000],
|
|
447
|
+
"call_summary": (getattr(result, "line", lambda: "")() or "")[:2000],
|
|
448
|
+
"transcript": segments_of(result),
|
|
449
|
+
"call_metadata": {
|
|
450
|
+
"harness_scenario": getattr(result, "scenario", ""),
|
|
451
|
+
"harness_passed": bool(getattr(result, "passed", False)),
|
|
452
|
+
"harness_met": int(getattr(result, "met", 0) or 0),
|
|
453
|
+
"harness_of": len(checkpoints),
|
|
454
|
+
"harness_checkpoints": checkpoints,
|
|
455
|
+
# Platform evaluations are backend-owned. Harness checks are direct
|
|
456
|
+
# execution evidence and stay namespaced in metadata rather than
|
|
457
|
+
# being sent through the removed SDK `evaluations` input field.
|
|
458
|
+
"harness_evaluations": evaluations,
|
|
459
|
+
"harness_eval_coverage": evaluation_coverage,
|
|
460
|
+
"harness_failure_classification": (
|
|
461
|
+
"grading_failure" if grading_failures else ""
|
|
462
|
+
),
|
|
463
|
+
"harness_spent_usd": round(
|
|
464
|
+
float(getattr(result, "spent_usd", 0.0) or 0.0), 4
|
|
465
|
+
),
|
|
466
|
+
},
|
|
467
|
+
}
|
|
468
|
+
if problems:
|
|
469
|
+
payload["error_message"] = "; ".join(problems)[:2000]
|
|
470
|
+
payload["result_digest"] = (
|
|
471
|
+
"sha256:"
|
|
472
|
+
+ hashlib.sha256(
|
|
473
|
+
json.dumps(
|
|
474
|
+
payload, sort_keys=True, separators=(",", ":"), default=str
|
|
475
|
+
).encode()
|
|
476
|
+
).hexdigest()
|
|
477
|
+
)
|
|
478
|
+
return payload
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def report(
|
|
482
|
+
results: list[Any],
|
|
483
|
+
scenarios: list[Any],
|
|
484
|
+
*,
|
|
485
|
+
name: str,
|
|
486
|
+
run_test_id: str = "",
|
|
487
|
+
modality: str = "text",
|
|
488
|
+
platform: Platform | None = None,
|
|
489
|
+
) -> Reported:
|
|
490
|
+
"""Report one suite run, and say where it landed.
|
|
491
|
+
|
|
492
|
+
``run_test_id`` is reused when the session already has one, so a second run adds a second
|
|
493
|
+
execution to the same test rather than a second test with one run in it.
|
|
494
|
+
|
|
495
|
+
``modality`` decides how the run is rendered: a spoken call reported as text lands in the
|
|
496
|
+
chat view, with no player and no audio, whatever actually happened on it.
|
|
497
|
+
"""
|
|
498
|
+
api = platform or Platform()
|
|
499
|
+
reported, ids = begin(
|
|
500
|
+
scenarios,
|
|
501
|
+
name=name,
|
|
502
|
+
run_test_id=run_test_id,
|
|
503
|
+
modality=modality,
|
|
504
|
+
platform=api,
|
|
505
|
+
)
|
|
506
|
+
|
|
507
|
+
# Calls come back in the order the scenarios were attached, which is the order they were run
|
|
508
|
+
# in. Zip rather than assume equal length: a suite can be a subset of its own test.
|
|
509
|
+
for call_execution_id, result in zip(ids, results, strict=False):
|
|
510
|
+
send_result(reported, call_execution_id, result, platform=api)
|
|
511
|
+
if len(ids) < len(results):
|
|
512
|
+
reported.problems.append(
|
|
513
|
+
f"the platform allocated {len(ids)} calls for {len(results)} scenarios, "
|
|
514
|
+
"so the rest were not reported"
|
|
515
|
+
)
|
|
516
|
+
return reported
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def begin(
|
|
520
|
+
scenarios: list[Any],
|
|
521
|
+
*,
|
|
522
|
+
name: str,
|
|
523
|
+
run_test_id: str = "",
|
|
524
|
+
modality: str = "text",
|
|
525
|
+
platform: Platform | None = None,
|
|
526
|
+
) -> tuple[Reported, list[str]]:
|
|
527
|
+
"""Create the platform rows before a suite starts, so the run is visible while it runs."""
|
|
528
|
+
api = platform or Platform()
|
|
529
|
+
reported = Reported(run_test_id=run_test_id)
|
|
530
|
+
provisioned_scenario_ids: list[str] = []
|
|
531
|
+
if not reported.run_test_id:
|
|
532
|
+
provisioned = api.provision(
|
|
533
|
+
name, [persona_of(one) for one in scenarios], modality=modality
|
|
534
|
+
)
|
|
535
|
+
reported.run_test_id = str(provisioned.get("run_test_id", ""))
|
|
536
|
+
# The provision endpoint returns IDs in the submitted persona order.
|
|
537
|
+
# Pass that order into execution creation; relying on a many-to-many
|
|
538
|
+
# queryset's database order can attach the right result to the wrong
|
|
539
|
+
# scenario row in the platform UI.
|
|
540
|
+
provisioned_scenario_ids = [
|
|
541
|
+
str(one) for one in provisioned.get("scenario_ids", [])
|
|
542
|
+
]
|
|
543
|
+
if not reported.run_test_id:
|
|
544
|
+
raise PlatformError("the platform returned no run test to report against")
|
|
545
|
+
|
|
546
|
+
harness_job_id = os.getenv("ALK_HARNESS_JOB_ID", "").strip()
|
|
547
|
+
selectors = [
|
|
548
|
+
{
|
|
549
|
+
"scenario_key": str(getattr(one, "name", "") or "")[:255],
|
|
550
|
+
"persona_name": str(persona_of(one).get("name") or "")[:255],
|
|
551
|
+
}
|
|
552
|
+
for one in scenarios
|
|
553
|
+
]
|
|
554
|
+
selector_kwargs = {"scenario_selectors": selectors} if selectors else {}
|
|
555
|
+
if harness_job_id:
|
|
556
|
+
started = api.start(
|
|
557
|
+
reported.run_test_id,
|
|
558
|
+
provisioned_scenario_ids,
|
|
559
|
+
harness_job_id=harness_job_id,
|
|
560
|
+
**selector_kwargs,
|
|
561
|
+
)
|
|
562
|
+
else:
|
|
563
|
+
# Keep the long-standing Platform-compatible call shape for local SDK and
|
|
564
|
+
# third-party implementations. The hosted ownership reference is additive
|
|
565
|
+
# and only exists inside a sandbox worker.
|
|
566
|
+
started = api.start(
|
|
567
|
+
reported.run_test_id,
|
|
568
|
+
provisioned_scenario_ids,
|
|
569
|
+
**selector_kwargs,
|
|
570
|
+
)
|
|
571
|
+
reported.test_execution_id = str(started.get("test_execution_id", ""))
|
|
572
|
+
if not reported.test_execution_id:
|
|
573
|
+
raise PlatformError("the platform returned no test execution for this run")
|
|
574
|
+
claimed = api.batch(reported.test_execution_id, max(1, len(scenarios)))
|
|
575
|
+
return reported, [str(one) for one in claimed.get("call_execution_ids", [])]
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def send_result(
|
|
579
|
+
reported: Reported,
|
|
580
|
+
call_execution_id: str,
|
|
581
|
+
result: Any,
|
|
582
|
+
*,
|
|
583
|
+
platform: Platform | None = None,
|
|
584
|
+
) -> None:
|
|
585
|
+
"""Patch one pre-allocated platform row as soon as its scenario finishes."""
|
|
586
|
+
api = platform or Platform()
|
|
587
|
+
try:
|
|
588
|
+
api.result(call_execution_id, result_of(result))
|
|
589
|
+
reported.calls[getattr(result, "scenario", "")] = call_execution_id
|
|
590
|
+
audio = str(getattr(result, "recording", "") or "")
|
|
591
|
+
if audio and Path(audio).exists():
|
|
592
|
+
try:
|
|
593
|
+
api.recording(call_execution_id, Path(audio))
|
|
594
|
+
except PlatformError as refused:
|
|
595
|
+
reported.problems.append(f"recording not sent: {refused}")
|
|
596
|
+
except PlatformError as failed:
|
|
597
|
+
reported.problems.append(f"{getattr(result, 'scenario', '?')}: {failed}")
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
def mark_ongoing(
|
|
601
|
+
reported: Reported,
|
|
602
|
+
call_execution_id: str,
|
|
603
|
+
*,
|
|
604
|
+
platform: Platform | None = None,
|
|
605
|
+
) -> None:
|
|
606
|
+
"""Best-effort PENDING -> ONGOING transition when a scenario actually starts.
|
|
607
|
+
|
|
608
|
+
A status ping is presentation state, not result evidence. The backend applies it only to a
|
|
609
|
+
pending call, so a duplicate or late ping cannot overwrite a terminal result. Failure here
|
|
610
|
+
must not fail the call or enter result-reconciliation bookkeeping.
|
|
611
|
+
"""
|
|
612
|
+
if not call_execution_id:
|
|
613
|
+
return
|
|
614
|
+
try:
|
|
615
|
+
(platform or Platform()).ongoing(call_execution_id)
|
|
616
|
+
except PlatformError:
|
|
617
|
+
pass
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def deliver(
|
|
621
|
+
results: list[Any],
|
|
622
|
+
scenarios: list[Any],
|
|
623
|
+
destination: Path | None,
|
|
624
|
+
*,
|
|
625
|
+
modality: str = "text",
|
|
626
|
+
) -> tuple[Reported | None, list[str]]:
|
|
627
|
+
"""Report a finished run, and say what happened, without ever failing the run.
|
|
628
|
+
|
|
629
|
+
Shared by every way a suite can be started, because a run that only appears on the platform
|
|
630
|
+
when it was started from one particular button is worse than one that never appears: which
|
|
631
|
+
runs exist then depends on how they were launched, and nobody can tell that from the page.
|
|
632
|
+
|
|
633
|
+
The suite has finished and its results are on disk by the time this is called, so an
|
|
634
|
+
unreachable platform is worth saying out loud and not worth throwing a completed run away
|
|
635
|
+
over. Returns what was reported, if anything, and the lines to show whoever asked.
|
|
636
|
+
"""
|
|
637
|
+
blocked = configured()
|
|
638
|
+
if blocked:
|
|
639
|
+
return None, [f"not reported to the platform: {blocked}"]
|
|
640
|
+
try:
|
|
641
|
+
reported = report(
|
|
642
|
+
results,
|
|
643
|
+
scenarios,
|
|
644
|
+
name=(destination.name if destination else "harness run"),
|
|
645
|
+
run_test_id=remembered(destination) if destination else "",
|
|
646
|
+
modality=modality,
|
|
647
|
+
)
|
|
648
|
+
except PlatformError as failed:
|
|
649
|
+
return None, [
|
|
650
|
+
f"the run finished, but reporting it to the platform failed: {failed}"
|
|
651
|
+
]
|
|
652
|
+
if destination:
|
|
653
|
+
remember(destination, reported)
|
|
654
|
+
said = [f"partly reported: {problem}" for problem in reported.problems]
|
|
655
|
+
said.append(f"reported to the platform: {reported.url}")
|
|
656
|
+
return reported, said
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def remember(destination: Path, reported: Reported) -> None:
|
|
660
|
+
"""Keep where a session reports to, so its next run joins the same test."""
|
|
661
|
+
(Path(destination) / "platform.json").write_text(
|
|
662
|
+
json.dumps(
|
|
663
|
+
{
|
|
664
|
+
"run_test_id": reported.run_test_id,
|
|
665
|
+
"test_execution_id": reported.test_execution_id,
|
|
666
|
+
"url": reported.url,
|
|
667
|
+
},
|
|
668
|
+
indent=2,
|
|
669
|
+
),
|
|
670
|
+
encoding="utf-8",
|
|
671
|
+
)
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def reported_to(destination: Path | None) -> dict[str, str]:
|
|
675
|
+
"""Where this session's runs have been reported, or nothing.
|
|
676
|
+
|
|
677
|
+
Read rather than held in memory, because a session outlives the process that reported it:
|
|
678
|
+
reopening one has to be able to find the run it already has.
|
|
679
|
+
"""
|
|
680
|
+
kept = Path(destination) / "platform.json" if destination else None
|
|
681
|
+
if kept is None or not kept.exists():
|
|
682
|
+
return {}
|
|
683
|
+
try:
|
|
684
|
+
found = json.loads(kept.read_text(encoding="utf-8"))
|
|
685
|
+
except Exception: # noqa: BLE001 - a damaged file just means provisioning again
|
|
686
|
+
return {}
|
|
687
|
+
return found if isinstance(found, dict) else {}
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def remembered(destination: Path) -> str:
|
|
691
|
+
"""The run test this session already has, or an empty string."""
|
|
692
|
+
return str(reported_to(destination).get("run_test_id", ""))
|