agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""Run the established ALK authoring stages for one hosted job.
|
|
2
|
+
|
|
3
|
+
This is intentionally a thin process boundary over :func:`fi.alk.harness.cli._auto`.
|
|
4
|
+
Contract creation, logical environment creation, and scenario generation therefore remain the
|
|
5
|
+
same implementation used by the local SDK and sandbox flows. Daytona consumes the frozen output
|
|
6
|
+
afterward; this command never executes scenarios itself.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import asyncio
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
import tempfile
|
|
17
|
+
|
|
18
|
+
from .cli import _auto
|
|
19
|
+
from .job import HarnessJob, ProviderExecutionMode, SourceKind
|
|
20
|
+
from .provider_import import inspect_provider_target
|
|
21
|
+
from .scenarios import load as load_written
|
|
22
|
+
from .understand import PROVIDER_IMPORT_PROFILE_PATH_ENV
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _persist_authored_scenario_count(
|
|
26
|
+
job_path: Path, job: HarnessJob, output: Path
|
|
27
|
+
) -> None:
|
|
28
|
+
"""Keep the frozen job in sync with adjustments applied during authoring.
|
|
29
|
+
|
|
30
|
+
The control plane may increase ``scenario_count`` while this process is already
|
|
31
|
+
running. ``_auto`` sees that adjustment and writes the larger validated suite,
|
|
32
|
+
but the following Bundle V2 process reloads this on-disk job document. Without
|
|
33
|
+
reconciling it here the bundler copies the original number of scenarios and the
|
|
34
|
+
platform correctly rejects preallocation because its expected count is newer.
|
|
35
|
+
"""
|
|
36
|
+
authored_count = len(load_written(output))
|
|
37
|
+
if authored_count <= 0 or authored_count == job.scenario_count:
|
|
38
|
+
return
|
|
39
|
+
updated = job.model_copy(update={"scenario_count": authored_count})
|
|
40
|
+
temporary = job_path.with_suffix(f"{job_path.suffix}.tmp")
|
|
41
|
+
temporary.write_text(
|
|
42
|
+
json.dumps(updated.model_dump(mode="json"), indent=2, sort_keys=True) + "\n",
|
|
43
|
+
encoding="utf-8",
|
|
44
|
+
)
|
|
45
|
+
temporary.replace(job_path)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _load_provider_import_profile(
|
|
49
|
+
job: HarnessJob,
|
|
50
|
+
secrets_path: Path | None,
|
|
51
|
+
profile_cache_path: Path | None = None,
|
|
52
|
+
) -> dict[str, object] | None:
|
|
53
|
+
inspect_connect_only_provider = (
|
|
54
|
+
job.agent.mode is ProviderExecutionMode.CONNECT_ONLY
|
|
55
|
+
and job.agent.connector.strip().lower() in {"vapi", "retell", "retell_chat"}
|
|
56
|
+
)
|
|
57
|
+
if (
|
|
58
|
+
job.agent.mode is not ProviderExecutionMode.PROVIDER_IMPORT
|
|
59
|
+
and not inspect_connect_only_provider
|
|
60
|
+
):
|
|
61
|
+
return None
|
|
62
|
+
if profile_cache_path is not None and profile_cache_path.is_file():
|
|
63
|
+
cached = json.loads(profile_cache_path.read_text(encoding="utf-8"))
|
|
64
|
+
if not isinstance(cached, dict):
|
|
65
|
+
raise RuntimeError("provider_import_authoring_profile_invalid")
|
|
66
|
+
return cached
|
|
67
|
+
if secrets_path is None:
|
|
68
|
+
raise RuntimeError("provider_import_authoring_secrets_missing")
|
|
69
|
+
try:
|
|
70
|
+
values = json.loads(secrets_path.read_text(encoding="utf-8"))
|
|
71
|
+
finally:
|
|
72
|
+
# The provider credential is needed only for this read-only inspection. Remove the file
|
|
73
|
+
# before any model session or source/environment process starts.
|
|
74
|
+
secrets_path.unlink(missing_ok=True)
|
|
75
|
+
if not isinstance(values, dict):
|
|
76
|
+
raise RuntimeError("provider_import_authoring_secrets_invalid")
|
|
77
|
+
connector = job.agent.connector.strip().lower()
|
|
78
|
+
provider = "retell" if connector == "retell_chat" else connector
|
|
79
|
+
secret_name = "VAPI_API_KEY" if provider == "vapi" else "RETELL_API_KEY"
|
|
80
|
+
target_key = "assistant_id" if provider == "vapi" else "agent_id"
|
|
81
|
+
profile = inspect_provider_target(
|
|
82
|
+
provider,
|
|
83
|
+
source_target_id=str(job.agent.config.get(target_key) or ""),
|
|
84
|
+
api_key=str(values.get(secret_name) or ""),
|
|
85
|
+
api_base_url=str(job.agent.config.get("provider_api_base_url") or "") or None,
|
|
86
|
+
target_modality="chat" if connector == "retell_chat" else "voice",
|
|
87
|
+
)
|
|
88
|
+
if profile_cache_path is not None:
|
|
89
|
+
profile_cache_path.write_text(
|
|
90
|
+
json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
|
91
|
+
)
|
|
92
|
+
return profile
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def main(argv: list[str] | None = None, *, validate_runtime: bool = False) -> int:
|
|
96
|
+
parser = argparse.ArgumentParser()
|
|
97
|
+
parser.add_argument("job", type=Path)
|
|
98
|
+
parser.add_argument("--source", type=Path, required=True)
|
|
99
|
+
parser.add_argument("--output", type=Path, required=True)
|
|
100
|
+
parser.add_argument(
|
|
101
|
+
"--adjustments",
|
|
102
|
+
type=Path,
|
|
103
|
+
help="JSONL inbox for user corrections applied at safe stage boundaries",
|
|
104
|
+
)
|
|
105
|
+
parser.add_argument(
|
|
106
|
+
"--target-secrets",
|
|
107
|
+
type=Path,
|
|
108
|
+
help="One-shot control-process secrets used to inspect an imported provider target",
|
|
109
|
+
)
|
|
110
|
+
parser.add_argument(
|
|
111
|
+
"--provider-profile-cache",
|
|
112
|
+
type=Path,
|
|
113
|
+
help="Control-owned sanitized profile reused across authoring retries",
|
|
114
|
+
)
|
|
115
|
+
args = parser.parse_args(argv)
|
|
116
|
+
|
|
117
|
+
job = HarnessJob.model_validate(json.loads(args.job.read_text(encoding="utf-8")))
|
|
118
|
+
profile = _load_provider_import_profile(
|
|
119
|
+
job, args.target_secrets, args.provider_profile_cache
|
|
120
|
+
)
|
|
121
|
+
# Transport kinds such as ``archive`` and ``github`` describe how the platform acquired the
|
|
122
|
+
# source. Once extracted, they are repositories. A source-free connect-only provider is the
|
|
123
|
+
# exception: its fetched definition is the source of truth and must never be represented by
|
|
124
|
+
# the intentionally empty /work/source directory.
|
|
125
|
+
source_free_provider = (
|
|
126
|
+
job.source.kind is SourceKind.PROVIDER
|
|
127
|
+
and job.agent.mode is ProviderExecutionMode.CONNECT_ONLY
|
|
128
|
+
and profile is not None
|
|
129
|
+
)
|
|
130
|
+
source_kind = (
|
|
131
|
+
"provider"
|
|
132
|
+
if source_free_provider
|
|
133
|
+
else str(job.metadata.get("source_kind") or "repo")
|
|
134
|
+
)
|
|
135
|
+
if source_kind not in {"repo", "spec", "provider"}:
|
|
136
|
+
source_kind = "repo"
|
|
137
|
+
namespace = argparse.Namespace(
|
|
138
|
+
path=str(args.source.resolve()),
|
|
139
|
+
name=str(job.metadata.get("agent_name") or args.source.name),
|
|
140
|
+
kind=source_kind,
|
|
141
|
+
out=str(args.output.resolve()),
|
|
142
|
+
count=job.scenario_count,
|
|
143
|
+
model=None,
|
|
144
|
+
run_model=None,
|
|
145
|
+
job=job,
|
|
146
|
+
adjustments_path=str(args.adjustments) if args.adjustments else None,
|
|
147
|
+
authoring_only=True,
|
|
148
|
+
provider_profile=profile if source_free_provider else None,
|
|
149
|
+
)
|
|
150
|
+
previous_profile_path = os.environ.get(PROVIDER_IMPORT_PROFILE_PATH_ENV)
|
|
151
|
+
with tempfile.TemporaryDirectory(prefix="alk-provider-profile-") as temporary:
|
|
152
|
+
if profile is not None:
|
|
153
|
+
profile_path = Path(temporary) / "provider-import-profile.json"
|
|
154
|
+
profile_path.write_text(
|
|
155
|
+
json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
|
156
|
+
)
|
|
157
|
+
os.environ[PROVIDER_IMPORT_PROFILE_PATH_ENV] = str(profile_path)
|
|
158
|
+
try:
|
|
159
|
+
status = asyncio.run(_auto(namespace))
|
|
160
|
+
if status == 0 and validate_runtime:
|
|
161
|
+
from .authoring_runtime_validation import validate_and_repair
|
|
162
|
+
|
|
163
|
+
_persist_authored_scenario_count(args.job, job, args.output.resolve())
|
|
164
|
+
runtime_job = HarnessJob.model_validate_json(args.job.read_text())
|
|
165
|
+
if profile is not None:
|
|
166
|
+
(args.output / "provider-import-profile.json").write_text(
|
|
167
|
+
json.dumps(profile) + "\n", encoding="utf-8"
|
|
168
|
+
)
|
|
169
|
+
asyncio.run(
|
|
170
|
+
validate_and_repair(
|
|
171
|
+
runtime_job, args.source.resolve(), args.output.resolve()
|
|
172
|
+
)
|
|
173
|
+
)
|
|
174
|
+
finally:
|
|
175
|
+
if previous_profile_path is None:
|
|
176
|
+
os.environ.pop(PROVIDER_IMPORT_PROFILE_PATH_ENV, None)
|
|
177
|
+
else:
|
|
178
|
+
os.environ[PROVIDER_IMPORT_PROFILE_PATH_ENV] = previous_profile_path
|
|
179
|
+
if status == 0:
|
|
180
|
+
_persist_authored_scenario_count(args.job, job, args.output.resolve())
|
|
181
|
+
if profile is not None:
|
|
182
|
+
(args.output.resolve() / "provider-import-profile.json").write_text(
|
|
183
|
+
json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
|
184
|
+
)
|
|
185
|
+
return status
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
if __name__ == "__main__":
|
|
189
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"""Validate generated setup against the actual hosted runtime before accepting authoring.
|
|
2
|
+
|
|
3
|
+
This is setup proof, not a claim that a reference tool trajectory executed. Actual
|
|
4
|
+
agent tool execution remains evidence collected during the calls.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import asyncio
|
|
11
|
+
import json
|
|
12
|
+
import random
|
|
13
|
+
import tempfile
|
|
14
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class RuntimeValidationError(RuntimeError):
|
|
19
|
+
def __init__(self, phase: str, detail: str):
|
|
20
|
+
self.phase = phase
|
|
21
|
+
super().__init__(detail)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
async def validate_once(
|
|
25
|
+
job,
|
|
26
|
+
source: Path,
|
|
27
|
+
authoring: Path,
|
|
28
|
+
*,
|
|
29
|
+
secrets_path=Path("/run/futureagi/secrets.json"),
|
|
30
|
+
) -> int:
|
|
31
|
+
from . import outbound
|
|
32
|
+
from .bundle_author_v2 import author_bundle_v2
|
|
33
|
+
from .hosted_entrypoint import (
|
|
34
|
+
ProcessWorldFactory,
|
|
35
|
+
_resolve_hosted_public_url,
|
|
36
|
+
job_secret_purposes,
|
|
37
|
+
)
|
|
38
|
+
from .hosted_scheduler import _classify_ready, _run_phase
|
|
39
|
+
from .job import ProviderExecutionMode, SourceKind
|
|
40
|
+
from .process_preflight import preflight_bundle
|
|
41
|
+
from .process_runtime import ProcessRuntimeProvider
|
|
42
|
+
from .scenario_source import load_scenarios
|
|
43
|
+
from .source_data_invariants import author_invariants, check_invariants
|
|
44
|
+
|
|
45
|
+
external_provider = (
|
|
46
|
+
getattr(getattr(job, "source", None), "kind", None) is SourceKind.PROVIDER
|
|
47
|
+
and getattr(getattr(job, "agent", None), "mode", None)
|
|
48
|
+
is ProviderExecutionMode.CONNECT_ONLY
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
# The real execution consumes its credential file. Validation gets a private copy,
|
|
52
|
+
# with the same purpose map, so it cannot destroy the execution handoff.
|
|
53
|
+
with tempfile.TemporaryDirectory(
|
|
54
|
+
prefix="runtime-validation-", dir=authoring.parent
|
|
55
|
+
) as root:
|
|
56
|
+
work = Path(root)
|
|
57
|
+
secrets = work / "secrets.json"
|
|
58
|
+
secrets.write_bytes(secrets_path.read_bytes())
|
|
59
|
+
secrets.chmod(0o600)
|
|
60
|
+
secret_values = tuple(
|
|
61
|
+
str(value) for value in json.loads(secrets.read_text()).values()
|
|
62
|
+
)
|
|
63
|
+
capabilities = outbound.load_capabilities(unlink=False)
|
|
64
|
+
transport = outbound.RequestsTransport()
|
|
65
|
+
provider = ProcessRuntimeProvider(
|
|
66
|
+
secrets_path=secrets,
|
|
67
|
+
secret_purpose_map=job_secret_purposes(job),
|
|
68
|
+
user_resolver=lambda _name: None,
|
|
69
|
+
require_declared_user=False,
|
|
70
|
+
public_url_resolver=lambda port, ttl: _resolve_hosted_public_url(
|
|
71
|
+
capabilities, transport, port=port, expires_in_seconds=ttl
|
|
72
|
+
),
|
|
73
|
+
provider_attempt_id=capabilities.attempt_id,
|
|
74
|
+
provider_expires_at=capabilities.expires_at,
|
|
75
|
+
)
|
|
76
|
+
executor = ThreadPoolExecutor(
|
|
77
|
+
max_workers=1, thread_name_prefix="runtime-validation"
|
|
78
|
+
)
|
|
79
|
+
phase = "environment"
|
|
80
|
+
try:
|
|
81
|
+
bundle = work / "bundle"
|
|
82
|
+
manifest = await asyncio.to_thread(
|
|
83
|
+
author_bundle_v2,
|
|
84
|
+
source=source,
|
|
85
|
+
job=job,
|
|
86
|
+
authoring=authoring,
|
|
87
|
+
output=bundle,
|
|
88
|
+
)
|
|
89
|
+
preflight_bundle(
|
|
90
|
+
bundle, manifest, parallelism=1, secret_refs=job_secret_purposes(job)
|
|
91
|
+
)
|
|
92
|
+
runtimes = await provider.provision(
|
|
93
|
+
manifest,
|
|
94
|
+
source=source,
|
|
95
|
+
bundle_dir=bundle,
|
|
96
|
+
work_directory=work,
|
|
97
|
+
instances=1,
|
|
98
|
+
)
|
|
99
|
+
factory = ProcessWorldFactory(work)
|
|
100
|
+
runtime = runtimes[0]
|
|
101
|
+
phase = "scenarios"
|
|
102
|
+
scenarios = await asyncio.to_thread(load_scenarios, bundle)
|
|
103
|
+
if len(scenarios) != job.scenario_count:
|
|
104
|
+
raise RuntimeValidationError(
|
|
105
|
+
phase, "Runtime scenario count differs from the requested count"
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
async def check_setups(invariants):
|
|
109
|
+
failures = []
|
|
110
|
+
for scenario in scenarios:
|
|
111
|
+
try:
|
|
112
|
+
await check_setup(scenario, invariants)
|
|
113
|
+
except RuntimeValidationError as exc:
|
|
114
|
+
failures.append(str(exc))
|
|
115
|
+
if failures:
|
|
116
|
+
raise RuntimeValidationError(phase, "\n".join(failures))
|
|
117
|
+
|
|
118
|
+
async def check_setup(scenario, invariants):
|
|
119
|
+
await provider.reset(runtime, work_directory=work)
|
|
120
|
+
world = await factory.create(runtime, rng=random.Random(job.seed or 0))
|
|
121
|
+
for name, fn, target, timeout in (
|
|
122
|
+
("setup", scenario.setup, world, 30.0),
|
|
123
|
+
("ready", scenario.ready, world.read_only(), 15.0),
|
|
124
|
+
):
|
|
125
|
+
result = await _run_phase(
|
|
126
|
+
fn, target, timeout=timeout, phase=name, executor=executor
|
|
127
|
+
)
|
|
128
|
+
if result.failure:
|
|
129
|
+
raise RuntimeValidationError(
|
|
130
|
+
phase, f"{scenario.scenario_key}: {name}: {result.failure}"
|
|
131
|
+
)
|
|
132
|
+
if name == "setup":
|
|
133
|
+
try:
|
|
134
|
+
await check_invariants(
|
|
135
|
+
world.read_only(),
|
|
136
|
+
invariants,
|
|
137
|
+
scenario_key=scenario.scenario_key,
|
|
138
|
+
)
|
|
139
|
+
except Exception as exc:
|
|
140
|
+
raise RuntimeValidationError(
|
|
141
|
+
phase, f"{scenario.scenario_key}: source data: {exc}"
|
|
142
|
+
) from exc
|
|
143
|
+
if name == "ready":
|
|
144
|
+
verdict = _classify_ready(result.value)
|
|
145
|
+
if verdict.broken or not verdict.held:
|
|
146
|
+
raise RuntimeValidationError(
|
|
147
|
+
phase,
|
|
148
|
+
f"{scenario.scenario_key}: ready precondition did not hold",
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
# Collect all executable setup errors before spending a model review or
|
|
152
|
+
# a repair attempt. Each scenario still gets an independent clean world.
|
|
153
|
+
await check_setups([])
|
|
154
|
+
if external_provider:
|
|
155
|
+
# A connect-only provider owns its state and executes its tools outside
|
|
156
|
+
# this sandbox. There is no harness-owned source database to probe or
|
|
157
|
+
# seed, so source-data invariant review would invent a local environment.
|
|
158
|
+
print(
|
|
159
|
+
"runtime validation: external provider black-box mode; "
|
|
160
|
+
"skipping local source-data invariant review",
|
|
161
|
+
flush=True,
|
|
162
|
+
)
|
|
163
|
+
return len(scenarios)
|
|
164
|
+
phase = "environment"
|
|
165
|
+
await provider.reset(runtime, work_directory=work)
|
|
166
|
+
baseline = await factory.create(runtime, rng=random.Random(job.seed or 0))
|
|
167
|
+
print("runtime validation: reviewing source data invariants", flush=True)
|
|
168
|
+
invariants = await author_invariants(
|
|
169
|
+
source, authoring, baseline.read_only(), endpoints=runtime.endpoints
|
|
170
|
+
)
|
|
171
|
+
# Review probes may have effects; none belongs in the test baseline.
|
|
172
|
+
await provider.reset(runtime, work_directory=work)
|
|
173
|
+
baseline = await factory.create(runtime, rng=random.Random(job.seed or 0))
|
|
174
|
+
await check_invariants(baseline.read_only(), invariants)
|
|
175
|
+
phase = "scenarios"
|
|
176
|
+
if invariants:
|
|
177
|
+
await check_setups(invariants)
|
|
178
|
+
return len(scenarios)
|
|
179
|
+
except RuntimeValidationError as exc:
|
|
180
|
+
raise RuntimeValidationError(
|
|
181
|
+
exc.phase,
|
|
182
|
+
outbound.redact_outbound_text(
|
|
183
|
+
str(exc), extra_secret_values=secret_values
|
|
184
|
+
),
|
|
185
|
+
) from None
|
|
186
|
+
except Exception as exc:
|
|
187
|
+
if "CERTIFICATE_VERIFY_FAILED" in str(exc):
|
|
188
|
+
# Generated data cannot repair the infrastructure trust store.
|
|
189
|
+
phase = "infrastructure"
|
|
190
|
+
raise RuntimeValidationError(
|
|
191
|
+
phase,
|
|
192
|
+
outbound.redact_outbound_text(
|
|
193
|
+
f"{type(exc).__name__}: {exc}", extra_secret_values=secret_values
|
|
194
|
+
),
|
|
195
|
+
) from None
|
|
196
|
+
finally:
|
|
197
|
+
executor.shutdown(wait=False, cancel_futures=True)
|
|
198
|
+
await provider.close(work_directory=work)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
async def validate_and_repair(
|
|
202
|
+
job, source: Path, authoring: Path, *, validate=validate_once, repair=None
|
|
203
|
+
) -> None:
|
|
204
|
+
"""Two repairs per phase; environment repairs cannot exhaust setup's budget."""
|
|
205
|
+
if repair is None:
|
|
206
|
+
|
|
207
|
+
async def repair(phase, guidance):
|
|
208
|
+
from .cli import _build, _scenarios
|
|
209
|
+
|
|
210
|
+
if phase == "environment":
|
|
211
|
+
return await _build(
|
|
212
|
+
argparse.Namespace(
|
|
213
|
+
name=source.name,
|
|
214
|
+
path=str(source),
|
|
215
|
+
out=str(authoring),
|
|
216
|
+
interactive=False,
|
|
217
|
+
guidance=[guidance],
|
|
218
|
+
skip_source_provision=True,
|
|
219
|
+
external_runtime=(
|
|
220
|
+
job.source.kind.value == "provider"
|
|
221
|
+
and getattr(job.agent.mode, "value", None) == "connect_only"
|
|
222
|
+
),
|
|
223
|
+
)
|
|
224
|
+
)
|
|
225
|
+
return await _scenarios(
|
|
226
|
+
argparse.Namespace(
|
|
227
|
+
name=source.name,
|
|
228
|
+
out=str(authoring),
|
|
229
|
+
count=job.scenario_count,
|
|
230
|
+
interactive=False,
|
|
231
|
+
guidance=[guidance],
|
|
232
|
+
)
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
repairs = {"environment": 0, "scenarios": 0}
|
|
236
|
+
for attempt in range(5):
|
|
237
|
+
print(f"runtime validation: attempt {attempt + 1}/5", flush=True)
|
|
238
|
+
try:
|
|
239
|
+
count = await validate(job, source, authoring)
|
|
240
|
+
except RuntimeValidationError as exc:
|
|
241
|
+
print(f"runtime validation: {exc.phase}: {exc}", flush=True)
|
|
242
|
+
if exc.phase not in repairs or repairs[exc.phase] >= 2:
|
|
243
|
+
raise
|
|
244
|
+
repairs[exc.phase] += 1
|
|
245
|
+
guidance = (
|
|
246
|
+
"Actual hosted runtime validation failed. Repair the generated environment/data "
|
|
247
|
+
"or scenario setup using the submitted source as authority. Do not modify source, "
|
|
248
|
+
"disable database constraints, weaken checks or ready conditions, drop scenarios, "
|
|
249
|
+
"or replace tools with invented implementations. Preserve scenario count and intent. "
|
|
250
|
+
f"Diagnostic: {str(exc)[:4000]}"
|
|
251
|
+
)
|
|
252
|
+
if await repair(exc.phase, guidance):
|
|
253
|
+
raise
|
|
254
|
+
else:
|
|
255
|
+
(authoring / "runtime-validation.json").write_text(
|
|
256
|
+
json.dumps(
|
|
257
|
+
{
|
|
258
|
+
"status": "passed",
|
|
259
|
+
"attempts": attempt + 1,
|
|
260
|
+
"setup_ready_scenarios": count,
|
|
261
|
+
"reference_tools_proven": False,
|
|
262
|
+
},
|
|
263
|
+
indent=2,
|
|
264
|
+
)
|
|
265
|
+
+ "\n"
|
|
266
|
+
)
|
|
267
|
+
return
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Harness backends
|
|
2
|
+
|
|
3
|
+
A stage of this harness is a conversation: a system prompt, tools, a turn budget, and a loop
|
|
4
|
+
that feeds tool results back until the model stops. A backend is whoever runs that loop.
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
ALK_HARNESS=claude # default; Claude Code loop, exactly the pre-seam behaviour
|
|
8
|
+
ALK_HARNESS=vertex-gemini # Google's ADK against Vertex (location=global)
|
|
9
|
+
ALK_HARNESS_MODEL=... # optional; unset means the backend's own default
|
|
10
|
+
ALK_VERTEX_LOCATION=... # vertex-gemini only; defaults to global (Gemini 3.x lives there)
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## The seam
|
|
14
|
+
|
|
15
|
+
- `base.py` is the whole contract. `SessionSpec` is what a stage asks for; the reply
|
|
16
|
+
vocabulary (`SessionOpened`, `ModelReply`, `ToolReturned`, `StageDone`) is what a session
|
|
17
|
+
emits; `HarnessBackend` / `HarnessSession` are the two protocols a backend implements.
|
|
18
|
+
- Tools are declared once, neutrally (`tool`, `tool_server`), and each backend adapts them:
|
|
19
|
+
the Claude backend builds an in-process MCP server, the Gemini backend builds function
|
|
20
|
+
declarations. Gating follows the same split: hook-based on Claude, structural on Gemini
|
|
21
|
+
(an ungranted tool is never declared).
|
|
22
|
+
- `Stage` (in `session.py`) drives any backend and renders the replies into events. It never
|
|
23
|
+
names a vendor.
|
|
24
|
+
|
|
25
|
+
## Adding a backend
|
|
26
|
+
|
|
27
|
+
Write a module with a class exposing `name`, `default_model`, `can_drive(model)` and
|
|
28
|
+
`create(spec) -> HarnessSession`, where the session yields the neutral replies and ends every
|
|
29
|
+
exchange with a `StageDone`. Register it in `__init__.py` (or call `register` from anywhere).
|
|
30
|
+
Nothing else in the harness changes: every stage, gate, and artifact works as-is. That is the
|
|
31
|
+
slot a Bedrock, Azure, or Gemini-CLI backend drops into.
|
|
32
|
+
|
|
33
|
+
Backends load lazily, so one backend's SDK is never imported because a different one ran.
|
|
34
|
+
|
|
35
|
+
## What the Gemini backend supplies itself
|
|
36
|
+
|
|
37
|
+
The ADK owns the loop, tool execution, and session history; the backend only adapts a
|
|
38
|
+
``ToolSpec`` through ADK's ``BaseTool`` extension point and translates its event stream into
|
|
39
|
+
the neutral replies. Claude Code ships Read/Glob/Grep and an operator-question tool; ADK has
|
|
40
|
+
no coding-CLI file tools, so `files.py` implements the read-only file tools once for any
|
|
41
|
+
backend that needs them. AskUserQuestion is deliberately not declared on the Gemini backend
|
|
42
|
+
yet: unattended runs never call it, and declaring a tool the backend cannot answer would cost
|
|
43
|
+
the model a turn finding that out.
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Harness backends, selected by name.
|
|
2
|
+
|
|
3
|
+
``ALK_HARNESS`` picks the backend the way ``ALK_HARNESS_MODEL`` already picks the model. With
|
|
4
|
+
nothing set the choice is ``vertex-gemini``, so a machine holding only Google credentials runs
|
|
5
|
+
without being told to; ``ALK_HARNESS=claude`` selects the Claude Code loop instead.
|
|
6
|
+
|
|
7
|
+
Backends load lazily: choosing one never imports the other's SDK, so a deployment installs only
|
|
8
|
+
the provider it uses. A new backend is a module implementing ``HarnessBackend`` plus one
|
|
9
|
+
``register`` call, from anywhere; nothing else in the harness changes.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import os
|
|
15
|
+
from typing import Callable
|
|
16
|
+
|
|
17
|
+
from .base import (
|
|
18
|
+
ASK_TOOL,
|
|
19
|
+
FILE_TOOLS,
|
|
20
|
+
KNOWN_BUILTINS,
|
|
21
|
+
Call,
|
|
22
|
+
HarnessBackend,
|
|
23
|
+
HarnessSession,
|
|
24
|
+
ModelReply,
|
|
25
|
+
Say,
|
|
26
|
+
SessionOpened,
|
|
27
|
+
SessionSpec,
|
|
28
|
+
StageDone,
|
|
29
|
+
ToolReturned,
|
|
30
|
+
ToolServer,
|
|
31
|
+
ToolSpec,
|
|
32
|
+
qualified,
|
|
33
|
+
tool,
|
|
34
|
+
tool_server,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
"ASK_TOOL",
|
|
39
|
+
"FILE_TOOLS",
|
|
40
|
+
"KNOWN_BUILTINS",
|
|
41
|
+
"Call",
|
|
42
|
+
"HarnessBackend",
|
|
43
|
+
"HarnessSession",
|
|
44
|
+
"ModelReply",
|
|
45
|
+
"Say",
|
|
46
|
+
"SessionOpened",
|
|
47
|
+
"SessionSpec",
|
|
48
|
+
"StageDone",
|
|
49
|
+
"ToolReturned",
|
|
50
|
+
"ToolServer",
|
|
51
|
+
"ToolSpec",
|
|
52
|
+
"qualified",
|
|
53
|
+
"tool",
|
|
54
|
+
"tool_server",
|
|
55
|
+
"register",
|
|
56
|
+
"resolve",
|
|
57
|
+
"backend_names",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
DEFAULT_BACKEND = "vertex-gemini"
|
|
61
|
+
|
|
62
|
+
_LOADERS: dict[str, Callable[[], HarnessBackend]] = {}
|
|
63
|
+
_ALIASES = {
|
|
64
|
+
"gemini": "vertex-gemini",
|
|
65
|
+
"vertex_gemini": "vertex-gemini",
|
|
66
|
+
"vertexai-gemini": "vertex-gemini",
|
|
67
|
+
"claude-code": "claude",
|
|
68
|
+
}
|
|
69
|
+
_LIVE: dict[str, HarnessBackend] = {}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def register(name: str, loader: Callable[[], HarnessBackend]) -> None:
|
|
73
|
+
"""Make a backend selectable by name. Loader runs on first use, not at registration."""
|
|
74
|
+
_LOADERS[name] = loader
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _load_claude() -> HarnessBackend:
|
|
78
|
+
from .claude import ClaudeBackend
|
|
79
|
+
|
|
80
|
+
return ClaudeBackend()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _load_vertex_gemini() -> HarnessBackend:
|
|
84
|
+
from .vertex_gemini import VertexGeminiBackend
|
|
85
|
+
|
|
86
|
+
return VertexGeminiBackend()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
register("claude", _load_claude)
|
|
90
|
+
register("vertex-gemini", _load_vertex_gemini)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def backend_names() -> list[str]:
|
|
94
|
+
return sorted(_LOADERS)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def resolve(name: str | None = None) -> HarnessBackend:
|
|
98
|
+
"""The backend a run will use: the one named, or ALK_HARNESS, or the default.
|
|
99
|
+
|
|
100
|
+
An unknown name is a loud error naming what exists. Falling back silently would run a whole
|
|
101
|
+
suite on the wrong harness, which is only discovered from the bill.
|
|
102
|
+
"""
|
|
103
|
+
asked = (name or os.environ.get("ALK_HARNESS") or DEFAULT_BACKEND).strip().lower()
|
|
104
|
+
asked = _ALIASES.get(asked, asked)
|
|
105
|
+
if asked not in _LOADERS:
|
|
106
|
+
raise ValueError(
|
|
107
|
+
f"no harness backend named {asked!r}; installed backends: "
|
|
108
|
+
f"{', '.join(backend_names())}"
|
|
109
|
+
)
|
|
110
|
+
if asked not in _LIVE:
|
|
111
|
+
_LIVE[asked] = _LOADERS[asked]()
|
|
112
|
+
backend = _LIVE[asked]
|
|
113
|
+
# A named model that this backend cannot reach is a configuration mistake, and it is only
|
|
114
|
+
# visible here. Left to run, the provider rejects the model mid-stage and the failure reads
|
|
115
|
+
# as the harness having nothing to say rather than as the wrong pairing.
|
|
116
|
+
wanted = os.environ.get("ALK_HARNESS_MODEL", "").strip()
|
|
117
|
+
if wanted and not backend.can_drive(wanted):
|
|
118
|
+
raise ValueError(
|
|
119
|
+
f"harness backend {backend.name!r} cannot drive model {wanted!r}; "
|
|
120
|
+
f"its default is {backend.default_model!r}"
|
|
121
|
+
)
|
|
122
|
+
return backend
|