agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Resolve opaque secret references only at the worker boundary."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
|
|
8
|
+
from fi.simulate.runtime.spec import SecretRef
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class SecretResolutionError(RuntimeError):
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
_RUNTIME_CONFIGURATION_PREFIX = "ALK_RUNTIME_VALUE_"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def runtime_configuration_environment(values: Mapping[str, str]) -> dict[str, str]:
|
|
19
|
+
"""Namespace submitted-agent values away from runner/controller credentials."""
|
|
20
|
+
return {
|
|
21
|
+
f"{_RUNTIME_CONFIGURATION_PREFIX}{name}": value
|
|
22
|
+
for name, value in values.items()
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def runtime_configuration_value(
|
|
27
|
+
name: str,
|
|
28
|
+
*,
|
|
29
|
+
environment: Mapping[str, str] | None = None,
|
|
30
|
+
) -> str:
|
|
31
|
+
"""Read a submitted-agent value without falling back to a controller secret."""
|
|
32
|
+
source = environment if environment is not None else os.environ
|
|
33
|
+
namespaced = source.get(f"{_RUNTIME_CONFIGURATION_PREFIX}{name}", "")
|
|
34
|
+
if namespaced:
|
|
35
|
+
return namespaced
|
|
36
|
+
# The local CLI historically receives agent configuration directly from its shell. Hosted
|
|
37
|
+
# workers always define the names manifest (even when empty), so they never fall back to a
|
|
38
|
+
# runner credential with the same name.
|
|
39
|
+
if "ALK_RUNTIME_CONFIGURATION_NAMES" not in source:
|
|
40
|
+
return source.get(name, "")
|
|
41
|
+
return ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def resolve_worker_secrets(
|
|
45
|
+
references: Mapping[str, SecretRef],
|
|
46
|
+
*,
|
|
47
|
+
environment: Mapping[str, str] | None = None,
|
|
48
|
+
) -> dict[str, str]:
|
|
49
|
+
"""Resolve mounted references without persisting or logging their values.
|
|
50
|
+
|
|
51
|
+
In local development ``environment`` reads the developer shell. In a hosted provider,
|
|
52
|
+
the secret manager mounts only job-authorized keys into the worker supervisor and this
|
|
53
|
+
same adapter reads those mounts. The job and platform continue to carry references only.
|
|
54
|
+
"""
|
|
55
|
+
source = environment if environment is not None else os.environ
|
|
56
|
+
resolved: dict[str, str] = {}
|
|
57
|
+
for alias, reference in references.items():
|
|
58
|
+
if reference.manager not in {"environment", "futureagi", "mounted"}:
|
|
59
|
+
raise SecretResolutionError(
|
|
60
|
+
f"secret_manager_unsupported: {reference.manager}"
|
|
61
|
+
)
|
|
62
|
+
value = source.get(reference.key)
|
|
63
|
+
if not value:
|
|
64
|
+
raise SecretResolutionError(f"secret_reference_unavailable: {alias}")
|
|
65
|
+
resolved[str(alias)] = value
|
|
66
|
+
return resolved
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def worker_environment(
|
|
70
|
+
resolved: Mapping[str, str],
|
|
71
|
+
*,
|
|
72
|
+
runtime_configuration: Mapping[str, str] | None = None,
|
|
73
|
+
host_environment: Mapping[str, str] | None = None,
|
|
74
|
+
) -> dict[str, str]:
|
|
75
|
+
"""Create a least-privilege child environment instead of inheriting every host secret."""
|
|
76
|
+
host = host_environment if host_environment is not None else os.environ
|
|
77
|
+
allowed = {
|
|
78
|
+
"DOCKER_HOST",
|
|
79
|
+
"HOME",
|
|
80
|
+
"LANG",
|
|
81
|
+
"LC_ALL",
|
|
82
|
+
"NO_PROXY",
|
|
83
|
+
"PATH",
|
|
84
|
+
"PYTHONPATH",
|
|
85
|
+
"SSL_CERT_DIR",
|
|
86
|
+
"SSL_CERT_FILE",
|
|
87
|
+
"TMPDIR",
|
|
88
|
+
# Runner topology. Hosted workers need these to reach sibling services privately and to
|
|
89
|
+
# expose their per-call webhook, but submitted values must never control them.
|
|
90
|
+
"ALK_DOCKER_BIND_HOST",
|
|
91
|
+
"ALK_DOCKER_NETWORK",
|
|
92
|
+
"ALK_DOCKER_PUBLISHED_HOST",
|
|
93
|
+
"ALK_RUNNER_CONTAINER",
|
|
94
|
+
"HARNESS_WEBHOOK_HOST",
|
|
95
|
+
"HARNESS_WEBHOOK_PORT",
|
|
96
|
+
"HARNESS_WEBHOOK_URL",
|
|
97
|
+
"HARNESS_RUNTIME_WEBHOOK_URL",
|
|
98
|
+
"HARNESS_VOICE_INFRA_RETRIES",
|
|
99
|
+
# Runner-owned model configuration. Uploaded agent values with these names remain in the
|
|
100
|
+
# runtime namespace and cannot replace controller credentials.
|
|
101
|
+
"ALK_AGENT_MODEL",
|
|
102
|
+
# The backend choice travels with the model it names: without it the worker
|
|
103
|
+
# falls back to the default backend and hands it a model it cannot drive.
|
|
104
|
+
"ALK_HARNESS",
|
|
105
|
+
"ALK_HARNESS_MODEL",
|
|
106
|
+
"ALK_HARNESS_THINKING",
|
|
107
|
+
"ALK_JUDGE_MODEL",
|
|
108
|
+
"ALK_USER_MODEL",
|
|
109
|
+
# Backend-specific configuration. A hosted run pinned to a region falls back to
|
|
110
|
+
# global without this, which is a quiet change of provider endpoint.
|
|
111
|
+
"ALK_VERTEX_LOCATION",
|
|
112
|
+
# The simulated caller's voice. Without it every persona shares one voice and the run
|
|
113
|
+
# stops exercising voice variation, which it reports as a log line rather than a failure.
|
|
114
|
+
"CARTESIA_API_KEY",
|
|
115
|
+
"ANTHROPIC_MODEL",
|
|
116
|
+
"ANTHROPIC_VERTEX_PROJECT_ID",
|
|
117
|
+
"CLAUDE_CODE_USE_VERTEX",
|
|
118
|
+
"CLOUD_ML_REGION",
|
|
119
|
+
"GOOGLE_APPLICATION_CREDENTIALS",
|
|
120
|
+
"GOOGLE_CLOUD_PROJECT",
|
|
121
|
+
# Runner-owned result delivery credentials, mounted independently of customer source.
|
|
122
|
+
"HARNESS_PLATFORM_URL",
|
|
123
|
+
"HARNESS_PLATFORM_API_KEY",
|
|
124
|
+
"HARNESS_PLATFORM_SECRET_KEY",
|
|
125
|
+
"FI_BASE_URL",
|
|
126
|
+
"FI_API_KEY",
|
|
127
|
+
"FI_SECRET_KEY",
|
|
128
|
+
}
|
|
129
|
+
child = {name: value for name, value in host.items() if name in allowed}
|
|
130
|
+
reserved = {
|
|
131
|
+
"ALK_HARNESS",
|
|
132
|
+
"ALK_HARNESS_MODEL",
|
|
133
|
+
"ALK_HARNESS_THINKING",
|
|
134
|
+
"ALK_VERTEX_LOCATION",
|
|
135
|
+
"CARTESIA_API_KEY",
|
|
136
|
+
"ANTHROPIC_MODEL",
|
|
137
|
+
"ANTHROPIC_VERTEX_PROJECT_ID",
|
|
138
|
+
"CLAUDE_CODE_USE_VERTEX",
|
|
139
|
+
"CLOUD_ML_REGION",
|
|
140
|
+
"GOOGLE_APPLICATION_CREDENTIALS",
|
|
141
|
+
"GOOGLE_CLOUD_PROJECT",
|
|
142
|
+
"ALK_DOCKER_NETWORK",
|
|
143
|
+
"ALK_DOCKER_BIND_HOST",
|
|
144
|
+
"ALK_DOCKER_PUBLISHED_HOST",
|
|
145
|
+
"ALK_RUNNER_CONTAINER",
|
|
146
|
+
"HARNESS_WEBHOOK_HOST",
|
|
147
|
+
"HARNESS_WEBHOOK_PORT",
|
|
148
|
+
"HARNESS_WEBHOOK_URL",
|
|
149
|
+
"HARNESS_RUNTIME_WEBHOOK_URL",
|
|
150
|
+
"HARNESS_VOICE_INFRA_RETRIES",
|
|
151
|
+
"ALK_AGENT_MODEL",
|
|
152
|
+
"ALK_JUDGE_MODEL",
|
|
153
|
+
"ALK_USER_MODEL",
|
|
154
|
+
}
|
|
155
|
+
child.update(
|
|
156
|
+
{name: value for name, value in resolved.items() if name not in reserved}
|
|
157
|
+
)
|
|
158
|
+
child.update(runtime_configuration_environment(runtime_configuration or {}))
|
|
159
|
+
return child
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
__all__ = [
|
|
163
|
+
"SecretResolutionError",
|
|
164
|
+
"resolve_worker_secrets",
|
|
165
|
+
"runtime_configuration_environment",
|
|
166
|
+
"runtime_configuration_value",
|
|
167
|
+
"worker_environment",
|
|
168
|
+
]
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Known service semantics layered on top of provider-neutral Compose discovery.
|
|
2
|
+
|
|
3
|
+
Compose remains the source of truth for *what* runs. This catalog only supplies the small
|
|
4
|
+
amount of semantics Compose cannot express: the protocol associated with a port, conventional
|
|
5
|
+
configuration names, and a useful readiness path. Unknown services are still supported as TCP
|
|
6
|
+
capabilities; adding a profile improves ergonomics without changing the lifecycle machinery.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ServiceProfile:
|
|
16
|
+
kind: str
|
|
17
|
+
port: int
|
|
18
|
+
protocol: str
|
|
19
|
+
configuration_names: tuple[str, ...] = ()
|
|
20
|
+
readiness_path: str = ""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
_PROFILES = (
|
|
24
|
+
ServiceProfile("postgres", 5432, "postgres", ("DATABASE_URL", "POSTGRES_URL")),
|
|
25
|
+
ServiceProfile("mysql", 3306, "mysql", ("DATABASE_URL", "MYSQL_URL")),
|
|
26
|
+
ServiceProfile(
|
|
27
|
+
"clickhouse",
|
|
28
|
+
8123,
|
|
29
|
+
"clickhouse",
|
|
30
|
+
("CLICKHOUSE_URL", "CLICKHOUSE_HTTP_URL"),
|
|
31
|
+
"/ping",
|
|
32
|
+
),
|
|
33
|
+
ServiceProfile("clickhouse", 9000, "tcp", ("CLICKHOUSE_NATIVE_URL",)),
|
|
34
|
+
ServiceProfile("redis", 6379, "redis", ("REDIS_URL",)),
|
|
35
|
+
ServiceProfile("mongodb", 27017, "mongodb", ("MONGODB_URL", "MONGO_URL")),
|
|
36
|
+
ServiceProfile("rabbitmq", 5672, "amqp", ("AMQP_URL", "RABBITMQ_URL")),
|
|
37
|
+
ServiceProfile("rabbitmq", 15672, "http", ("RABBITMQ_MANAGEMENT_URL",)),
|
|
38
|
+
ServiceProfile("kafka", 9092, "kafka", ("KAFKA_BOOTSTRAP_SERVERS", "KAFKA_URL")),
|
|
39
|
+
ServiceProfile("nats", 4222, "nats", ("NATS_URL",)),
|
|
40
|
+
ServiceProfile("nats", 8222, "http", ("NATS_MONITORING_URL",)),
|
|
41
|
+
ServiceProfile(
|
|
42
|
+
"minio", 9000, "s3", ("S3_ENDPOINT_URL", "MINIO_URL"), "/minio/health/ready"
|
|
43
|
+
),
|
|
44
|
+
ServiceProfile("minio", 9001, "http", ("MINIO_CONSOLE_URL",)),
|
|
45
|
+
ServiceProfile(
|
|
46
|
+
"elasticsearch", 9200, "http", ("ELASTICSEARCH_URL", "SEARCH_URL"), "/"
|
|
47
|
+
),
|
|
48
|
+
ServiceProfile("qdrant", 6333, "http", ("QDRANT_URL",), "/readyz"),
|
|
49
|
+
ServiceProfile("qdrant", 6334, "grpc", ("QDRANT_GRPC_URL",)),
|
|
50
|
+
ServiceProfile("neo4j", 7474, "http", ("NEO4J_HTTP_URL",)),
|
|
51
|
+
ServiceProfile("neo4j", 7687, "bolt", ("NEO4J_URI", "NEO4J_URL")),
|
|
52
|
+
ServiceProfile("livekit", 7880, "livekit", ("LIVEKIT_URL",)),
|
|
53
|
+
ServiceProfile(
|
|
54
|
+
"code-executor", 8000, "http", ("CODE_EXECUTOR_URL",), "/health"
|
|
55
|
+
),
|
|
56
|
+
ServiceProfile("mcp", 8000, "mcp", ("MCP_URL", "MCP_SERVER_URL")),
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
_ALIASES = {
|
|
61
|
+
"mongo": "mongodb",
|
|
62
|
+
"opensearch": "elasticsearch",
|
|
63
|
+
"redpanda": "kafka",
|
|
64
|
+
"seaweedfs": "s3",
|
|
65
|
+
"code_executor": "code-executor",
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def profile_for(service: str, image: str, port: int) -> ServiceProfile:
|
|
70
|
+
"""Return semantics for one endpoint, retaining unknown endpoints as TCP."""
|
|
71
|
+
haystack = f"{service} {image}".lower()
|
|
72
|
+
for profile in _PROFILES:
|
|
73
|
+
names = {
|
|
74
|
+
profile.kind,
|
|
75
|
+
*[key for key, value in _ALIASES.items() if value == profile.kind],
|
|
76
|
+
}
|
|
77
|
+
if profile.port == port and any(name in haystack for name in names):
|
|
78
|
+
return profile
|
|
79
|
+
# Port is a strong signal even when a private image has an opaque registry/name.
|
|
80
|
+
candidates = [profile for profile in _PROFILES if profile.port == port]
|
|
81
|
+
if len(candidates) == 1:
|
|
82
|
+
return candidates[0]
|
|
83
|
+
return ServiceProfile("service", port, "tcp")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def address(protocol: str, host: str, port: int) -> str:
|
|
87
|
+
"""Canonical non-secret connector value for a discovered endpoint."""
|
|
88
|
+
scheme = {
|
|
89
|
+
"clickhouse": "http",
|
|
90
|
+
"s3": "http",
|
|
91
|
+
"mcp": "http",
|
|
92
|
+
"livekit": "ws",
|
|
93
|
+
}.get(protocol, protocol)
|
|
94
|
+
return f"{scheme}://{host}:{port}"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
__all__ = ["ServiceProfile", "address", "profile_for"]
|
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""A stage as a live conversation, emitting what happened as it happens.
|
|
2
|
+
|
|
3
|
+
The operator experiences one continuous session: point at an agent, watch a contract appear,
|
|
4
|
+
correct something, move on. Underneath, each stage is its own session so context stays small and
|
|
5
|
+
any stage can be re-entered without redoing the ones before it.
|
|
6
|
+
|
|
7
|
+
A stage stays open across turns, so a correction is the next thing said rather than a re-run,
|
|
8
|
+
and it yields typed events rather than a wall of text. A terminal renders those events as lines;
|
|
9
|
+
a browser renders the same events as a transcript on one side and the artifact on the other.
|
|
10
|
+
Neither is privileged, which is the point.
|
|
11
|
+
|
|
12
|
+
Which loop actually runs the conversation is a backend, selected by ``ALK_HARNESS`` through
|
|
13
|
+
``backends.resolve``. A stage describes what it needs in a ``SessionSpec``; the backend supplies
|
|
14
|
+
the session and translates its provider's stream into the small reply vocabulary this module
|
|
15
|
+
renders. Nothing above this line knows a vendor's name.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import asyncio
|
|
21
|
+
import os
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from typing import Any, AsyncIterator, Callable
|
|
24
|
+
|
|
25
|
+
from . import spend
|
|
26
|
+
from .backends import (
|
|
27
|
+
Call,
|
|
28
|
+
HarnessBackend,
|
|
29
|
+
HarnessSession,
|
|
30
|
+
ModelReply,
|
|
31
|
+
Say,
|
|
32
|
+
SessionOpened,
|
|
33
|
+
SessionSpec,
|
|
34
|
+
StageDone,
|
|
35
|
+
ToolReturned,
|
|
36
|
+
ToolServer,
|
|
37
|
+
resolve,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
TEXT = "text"
|
|
41
|
+
TOOL = "tool"
|
|
42
|
+
RESULT = "result"
|
|
43
|
+
ARTIFACT = "artifact"
|
|
44
|
+
DONE = "done"
|
|
45
|
+
|
|
46
|
+
# Provider streams normally emit a message or tool event every few seconds. A subprocess can
|
|
47
|
+
# remain alive forever after a dropped upstream stream, though, which previously left a hosted
|
|
48
|
+
# job looking healthy while making no progress. Bound *inactivity*, not total stage duration:
|
|
49
|
+
# long scenario suites remain valid as long as they keep producing observable work.
|
|
50
|
+
DEFAULT_STAGE_IDLE_TIMEOUT_SECONDS = 600.0
|
|
51
|
+
STAGE_IDLE_TIMEOUT_SECONDS = float(
|
|
52
|
+
os.getenv("ALK_STAGE_IDLE_TIMEOUT_SECONDS", str(DEFAULT_STAGE_IDLE_TIMEOUT_SECONDS))
|
|
53
|
+
)
|
|
54
|
+
STAGE_IDLE_RETRIES = int(os.getenv("ALK_STAGE_IDLE_RETRIES", "1"))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class StageIdleTimeout(TimeoutError):
|
|
58
|
+
"""The provider stream stayed open without producing any observable event."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class Event:
|
|
63
|
+
"""One observable thing the stage did.
|
|
64
|
+
|
|
65
|
+
``detail`` carries the data behind what is being shown, not just a label for it: which stage
|
|
66
|
+
emitted this, and for a tool call the arguments it was made with. A terminal renders a line
|
|
67
|
+
and ignores the rest; anything richer needs the data, and re-parsing a rendered line to get
|
|
68
|
+
it back is how a second front end becomes a rewrite.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
kind: str
|
|
72
|
+
text: str = ""
|
|
73
|
+
tool: str = ""
|
|
74
|
+
detail: dict[str, Any] = field(default_factory=dict)
|
|
75
|
+
|
|
76
|
+
def line(self) -> str:
|
|
77
|
+
"""A terminal-friendly rendering."""
|
|
78
|
+
if self.kind == TEXT:
|
|
79
|
+
return self.text
|
|
80
|
+
if self.kind == TOOL:
|
|
81
|
+
target = self.detail.get("target") or ""
|
|
82
|
+
return f" [{self.tool}{' ' + target if target else ''}]"
|
|
83
|
+
if self.kind == RESULT:
|
|
84
|
+
marker = "!" if self.detail.get("is_error") else ">"
|
|
85
|
+
body = "\n".join(
|
|
86
|
+
f" {marker} {row}" for row in self.text.splitlines() if row
|
|
87
|
+
)
|
|
88
|
+
return body or f" {marker} (no output)"
|
|
89
|
+
if self.kind == ARTIFACT:
|
|
90
|
+
return f" [saved {self.detail.get('path', '')}]"
|
|
91
|
+
if self.kind == DONE:
|
|
92
|
+
cost = self.detail.get("cost_usd")
|
|
93
|
+
spent = f" ${cost:.4f}" if isinstance(cost, float) else ""
|
|
94
|
+
failure = self.detail.get("error")
|
|
95
|
+
wrong = self.detail.get("unexpected_model") or []
|
|
96
|
+
return (
|
|
97
|
+
f" [{self.detail.get('outcome', '')} "
|
|
98
|
+
f"turns={self.detail.get('turns', 0)}{spent}]"
|
|
99
|
+
+ (f"\n !! {failure}" if failure else "")
|
|
100
|
+
+ (
|
|
101
|
+
f"\n !! billed to {', '.join(wrong)}, which is not what was asked for"
|
|
102
|
+
if wrong
|
|
103
|
+
else ""
|
|
104
|
+
)
|
|
105
|
+
)
|
|
106
|
+
return self.text
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass
|
|
110
|
+
class Turn:
|
|
111
|
+
"""What one exchange produced."""
|
|
112
|
+
|
|
113
|
+
text: str = ""
|
|
114
|
+
events: list[Event] = field(default_factory=list)
|
|
115
|
+
tools_used: list[str] = field(default_factory=list)
|
|
116
|
+
artifacts: list[str] = field(default_factory=list)
|
|
117
|
+
outcome: str = ""
|
|
118
|
+
turns: int = 0
|
|
119
|
+
cost_usd: float | None = None
|
|
120
|
+
error: str = ""
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
_TARGET_KEYS = (
|
|
124
|
+
"file_path",
|
|
125
|
+
"path",
|
|
126
|
+
"pattern",
|
|
127
|
+
"agent",
|
|
128
|
+
"tool",
|
|
129
|
+
"tool_name",
|
|
130
|
+
"table",
|
|
131
|
+
"name",
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _why_it_failed(done: StageDone) -> str:
|
|
136
|
+
"""What actually went wrong, said in terms somebody can act on."""
|
|
137
|
+
said = "; ".join(str(error) for error in done.errors)[:400]
|
|
138
|
+
if "invalid_rapt" in said or "invalid_grant" in said:
|
|
139
|
+
return (
|
|
140
|
+
"the provider rejected the credentials. GOOGLE_APPLICATION_CREDENTIALS is probably "
|
|
141
|
+
"not set in this shell, so it fell back to your gcloud login. Load the env file "
|
|
142
|
+
"first: set -a; . ./.env.acceptance; set +a"
|
|
143
|
+
)
|
|
144
|
+
status = done.api_error_status
|
|
145
|
+
return f"the model call failed{f' ({status})' if status else ''}: {said or 'no detail given'}"
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def readable(tool_name: str) -> str:
|
|
149
|
+
"""A tool's name as somebody reading along would say it.
|
|
150
|
+
|
|
151
|
+
``mcp__scenarios__try_calls`` is how the model addresses it and is noise to anybody else.
|
|
152
|
+
"""
|
|
153
|
+
bare = tool_name.rsplit("__", 1)[-1]
|
|
154
|
+
return bare.replace("_", " ")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _target(payload: Any) -> str:
|
|
158
|
+
"""A short label for what a tool call was aimed at, for display only."""
|
|
159
|
+
if not isinstance(payload, dict):
|
|
160
|
+
return ""
|
|
161
|
+
for key in _TARGET_KEYS:
|
|
162
|
+
value = payload.get(key)
|
|
163
|
+
if isinstance(value, str) and value:
|
|
164
|
+
return value if len(value) <= 80 else value[:77] + "..."
|
|
165
|
+
return ""
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _shown(text: str, limit: int = 600) -> str:
|
|
169
|
+
return text if len(text) <= limit else text[: limit - 3] + "..."
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _saved_path(text: str) -> str:
|
|
173
|
+
"""The path a tool reports having written, if it wrote one.
|
|
174
|
+
|
|
175
|
+
Only when the tool actually says it saved something. Matching any path-shaped token in any
|
|
176
|
+
result meant that reading a file announced it as an artifact — the stage looks like it is
|
|
177
|
+
producing output while it is still only looking around, and a front end reloads its panes on
|
|
178
|
+
every read.
|
|
179
|
+
"""
|
|
180
|
+
said = text.lower()
|
|
181
|
+
if not any(verb in said for verb in ("saved", "wrote", "written")):
|
|
182
|
+
return ""
|
|
183
|
+
for token in text.split():
|
|
184
|
+
# Trimmed before the check, not after. A tool that ends its sentence — "saved to
|
|
185
|
+
# out/contract.json." — produces a token ending in the full stop, so testing the
|
|
186
|
+
# suffix first missed every real save and matched only bare paths, which is what a
|
|
187
|
+
# file *read* returns. The event fired on exactly the wrong occasions.
|
|
188
|
+
cleaned = token.strip(".,;:!?)\"'")
|
|
189
|
+
if cleaned.endswith((".json", ".py", ".sqlite")):
|
|
190
|
+
return cleaned
|
|
191
|
+
return ""
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
class Stage:
|
|
195
|
+
"""One stage of the harness, held open so it can be talked to."""
|
|
196
|
+
|
|
197
|
+
def __init__(
|
|
198
|
+
self,
|
|
199
|
+
spec: SessionSpec,
|
|
200
|
+
*,
|
|
201
|
+
name: str = "",
|
|
202
|
+
backend: HarnessBackend | None = None,
|
|
203
|
+
) -> None:
|
|
204
|
+
self._spec = spec
|
|
205
|
+
self._backend = backend
|
|
206
|
+
self._session: HarnessSession | None = None
|
|
207
|
+
self.name = name
|
|
208
|
+
self.session_id: str | None = None
|
|
209
|
+
self.history: list[Turn] = []
|
|
210
|
+
# What actually got billed, read back rather than assumed. Asking for a model is not the
|
|
211
|
+
# same as getting one: a request that quietly does not take shows up only on the
|
|
212
|
+
# invoice, weeks later, as a number nobody can explain.
|
|
213
|
+
self.models_used: set[str] = set()
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def spec(self) -> SessionSpec:
|
|
217
|
+
return self._spec
|
|
218
|
+
|
|
219
|
+
def grant(
|
|
220
|
+
self, server_name: str, server: ToolServer, tool_names: list[str], ask: Any = None
|
|
221
|
+
) -> None:
|
|
222
|
+
"""Give this stage one more tool server, before it opens.
|
|
223
|
+
|
|
224
|
+
The backend builds its permission surface from the spec when the session opens, so a
|
|
225
|
+
grant is a spec change and must land before then. ``tool_names`` is accepted for
|
|
226
|
+
compatibility with existing callers; the server's own tool list is authoritative.
|
|
227
|
+
``ask`` replaces the operator callback when given, as it always has.
|
|
228
|
+
"""
|
|
229
|
+
if self._session is not None:
|
|
230
|
+
raise RuntimeError(
|
|
231
|
+
"grant before the stage opens; the session is already running"
|
|
232
|
+
)
|
|
233
|
+
del tool_names
|
|
234
|
+
self._spec.grant(server_name, server)
|
|
235
|
+
if ask is not None:
|
|
236
|
+
self._spec.ask = ask
|
|
237
|
+
|
|
238
|
+
async def __aenter__(self) -> "Stage":
|
|
239
|
+
if self._backend is None:
|
|
240
|
+
self._backend = resolve()
|
|
241
|
+
self._session = self._backend.create(self._spec)
|
|
242
|
+
await self._session.start()
|
|
243
|
+
return self
|
|
244
|
+
|
|
245
|
+
async def __aexit__(self, *_exc: Any) -> None:
|
|
246
|
+
if self._session is not None:
|
|
247
|
+
await self._session.stop()
|
|
248
|
+
self._session = None
|
|
249
|
+
|
|
250
|
+
async def stream(self, message: str) -> AsyncIterator[Event]:
|
|
251
|
+
"""Send a message and yield events as they arrive."""
|
|
252
|
+
if self._session is None:
|
|
253
|
+
raise RuntimeError("stage is not open; use it as an async context manager")
|
|
254
|
+
await self._session.send(message)
|
|
255
|
+
turn = Turn()
|
|
256
|
+
replies = self._session.replies().__aiter__()
|
|
257
|
+
# A stage may ask for a longer silence than the default, because for some stages silence
|
|
258
|
+
# is the work: a session whose turn is one tool call that fans out to other sessions
|
|
259
|
+
# emits nothing until that call returns, and killing it then throws away everything the
|
|
260
|
+
# delegates proved.
|
|
261
|
+
idle_bound = self._spec.idle_timeout_seconds or STAGE_IDLE_TIMEOUT_SECONDS
|
|
262
|
+
while True:
|
|
263
|
+
try:
|
|
264
|
+
received = await asyncio.wait_for(
|
|
265
|
+
replies.__anext__(), timeout=idle_bound
|
|
266
|
+
)
|
|
267
|
+
except StopAsyncIteration:
|
|
268
|
+
break
|
|
269
|
+
except TimeoutError as exc:
|
|
270
|
+
raise StageIdleTimeout(
|
|
271
|
+
f"{self.name or 'model'} produced no event for {idle_bound:g}s"
|
|
272
|
+
) from exc
|
|
273
|
+
for event in self._events(received, turn):
|
|
274
|
+
# Which stage this came from, stamped once here rather than by every caller,
|
|
275
|
+
# so a front end showing several stages can tell them apart.
|
|
276
|
+
event.detail.setdefault("stage", self.name)
|
|
277
|
+
turn.events.append(event)
|
|
278
|
+
yield event
|
|
279
|
+
self.history.append(turn)
|
|
280
|
+
|
|
281
|
+
def _events(self, received: Any, turn: Turn) -> list[Event]:
|
|
282
|
+
if isinstance(received, SessionOpened):
|
|
283
|
+
self.session_id = received.session_id or self.session_id
|
|
284
|
+
return []
|
|
285
|
+
if isinstance(received, ModelReply):
|
|
286
|
+
events: list[Event] = []
|
|
287
|
+
for part in received.parts:
|
|
288
|
+
if isinstance(part, Say):
|
|
289
|
+
turn.text += part.text
|
|
290
|
+
events.append(Event(TEXT, text=part.text))
|
|
291
|
+
elif isinstance(part, Call):
|
|
292
|
+
turn.tools_used.append(part.name)
|
|
293
|
+
events.append(
|
|
294
|
+
Event(
|
|
295
|
+
TOOL,
|
|
296
|
+
tool=part.name,
|
|
297
|
+
detail={
|
|
298
|
+
"target": _target(part.arguments),
|
|
299
|
+
"arguments": part.arguments,
|
|
300
|
+
"label": readable(part.name),
|
|
301
|
+
},
|
|
302
|
+
)
|
|
303
|
+
)
|
|
304
|
+
return events
|
|
305
|
+
if isinstance(received, ToolReturned):
|
|
306
|
+
# What a tool said back is the only view a caller has of whether the work is
|
|
307
|
+
# going well. Dropping it leaves a run that can only be diagnosed by guessing.
|
|
308
|
+
events = [
|
|
309
|
+
Event(
|
|
310
|
+
RESULT,
|
|
311
|
+
text=_shown(received.text),
|
|
312
|
+
detail={"is_error": received.is_error},
|
|
313
|
+
)
|
|
314
|
+
]
|
|
315
|
+
path = _saved_path(received.text)
|
|
316
|
+
if path:
|
|
317
|
+
turn.artifacts.append(path)
|
|
318
|
+
events.append(Event(ARTIFACT, detail={"path": path}))
|
|
319
|
+
return events
|
|
320
|
+
if isinstance(received, StageDone):
|
|
321
|
+
# The reported outcome alone is not the outcome. A call that failed upstream can
|
|
322
|
+
# still arrive saying "success", so reporting it verbatim tells somebody their
|
|
323
|
+
# stage worked when nothing happened at all.
|
|
324
|
+
failed = bool(received.is_error or received.api_error_status)
|
|
325
|
+
turn.outcome = "failed" if failed else received.outcome
|
|
326
|
+
turn.turns = received.turns
|
|
327
|
+
turn.cost_usd = received.cost_usd
|
|
328
|
+
# Every harness model call passes here, so the spend ledger is fed once rather than
|
|
329
|
+
# per stage: a writer added later is counted without anybody remembering to.
|
|
330
|
+
spend.record(
|
|
331
|
+
self.name,
|
|
332
|
+
received.cost_usd,
|
|
333
|
+
received.turns,
|
|
334
|
+
received.models,
|
|
335
|
+
received.tokens_in,
|
|
336
|
+
received.tokens_out,
|
|
337
|
+
received.tokens_cached,
|
|
338
|
+
)
|
|
339
|
+
turn.error = _why_it_failed(received) if failed else ""
|
|
340
|
+
self.session_id = received.session_id or self.session_id
|
|
341
|
+
self.models_used |= received.models
|
|
342
|
+
unexpected = self.unexpected_models()
|
|
343
|
+
return [
|
|
344
|
+
Event(
|
|
345
|
+
DONE,
|
|
346
|
+
detail={
|
|
347
|
+
"outcome": turn.outcome,
|
|
348
|
+
"turns": received.turns,
|
|
349
|
+
"cost_usd": received.cost_usd,
|
|
350
|
+
"error": turn.error,
|
|
351
|
+
"models": sorted(received.models),
|
|
352
|
+
"unexpected_model": sorted(unexpected),
|
|
353
|
+
},
|
|
354
|
+
)
|
|
355
|
+
]
|
|
356
|
+
return []
|
|
357
|
+
|
|
358
|
+
async def say(
|
|
359
|
+
self, message: str, *, on_event: Callable[[Event], None] | None = None
|
|
360
|
+
) -> Turn:
|
|
361
|
+
"""Send a message and wait for the whole reply."""
|
|
362
|
+
for attempt in range(STAGE_IDLE_RETRIES + 1):
|
|
363
|
+
try:
|
|
364
|
+
async for event in self.stream(message):
|
|
365
|
+
if on_event:
|
|
366
|
+
on_event(event)
|
|
367
|
+
return self.history[-1]
|
|
368
|
+
except StageIdleTimeout:
|
|
369
|
+
if attempt >= STAGE_IDLE_RETRIES:
|
|
370
|
+
raise
|
|
371
|
+
# A timed-out receive has been cancelled and the provider session may still be
|
|
372
|
+
# waiting on a dead stream. Reusing it can only reproduce the dead stream.
|
|
373
|
+
# Start a clean session and replay the same stage instruction. Harness writes
|
|
374
|
+
# are named/idempotent and remain protected by their validation gates.
|
|
375
|
+
if self._session is not None:
|
|
376
|
+
await self._session.stop()
|
|
377
|
+
assert self._backend is not None
|
|
378
|
+
self._session = self._backend.create(self._spec)
|
|
379
|
+
await self._session.start()
|
|
380
|
+
raise AssertionError("unreachable")
|
|
381
|
+
|
|
382
|
+
def unexpected_models(self) -> set[str]:
|
|
383
|
+
"""Models that were billed but not the one asked for."""
|
|
384
|
+
asked = self._spec.model
|
|
385
|
+
if not asked:
|
|
386
|
+
return set()
|
|
387
|
+
return {used for used in self.models_used if asked.split("-2")[0] not in used}
|
|
388
|
+
|
|
389
|
+
@property
|
|
390
|
+
def spent_usd(self) -> float:
|
|
391
|
+
return sum(turn.cost_usd or 0.0 for turn in self.history)
|