agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/harness/config.py
ADDED
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
"""Session configuration for the harness.
|
|
2
|
+
|
|
3
|
+
One place decides which model runs, how the session reaches it, and what the agent is allowed to
|
|
4
|
+
touch. Every stage builds its options from here so that a change of provider or model is one
|
|
5
|
+
edit rather than a search across stages.
|
|
6
|
+
|
|
7
|
+
Credentials are never read from source. The Vertex project and credential path come from the
|
|
8
|
+
environment, which is also how the rest of the platform resolves them.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import os
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Iterable
|
|
16
|
+
|
|
17
|
+
from .backends import SessionSpec, resolve
|
|
18
|
+
|
|
19
|
+
# The first backend's default, kept importable because callers and tests name it. The model a
|
|
20
|
+
# run actually gets comes from chosen_model, which asks the selected backend.
|
|
21
|
+
DEFAULT_MODEL = "claude-sonnet-4-6"
|
|
22
|
+
|
|
23
|
+
SKILLS_ROOT = Path(__file__).parent / "skills"
|
|
24
|
+
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
|
25
|
+
ARTIFACTS_ROOT = PROJECT_ROOT / "artifacts"
|
|
26
|
+
|
|
27
|
+
_READ_ONLY_TOOLS = ("Read", "Glob", "Grep")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def credentials_hint() -> str:
|
|
31
|
+
"""A line saying which credentials a run will use, or a warning that it is guessing.
|
|
32
|
+
|
|
33
|
+
Claude Code falls back to the active gcloud login when no service-account file is named,
|
|
34
|
+
which is a legitimate setup and an easy accident. The accident produces a provider auth
|
|
35
|
+
error several layers down, so it is worth saying out loud which one is in play.
|
|
36
|
+
"""
|
|
37
|
+
named = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS")
|
|
38
|
+
if named:
|
|
39
|
+
return f"credentials: {Path(named).name}"
|
|
40
|
+
return (
|
|
41
|
+
"credentials: none named, falling back to your gcloud login. If calls fail to "
|
|
42
|
+
"authenticate, load the env file first:\n"
|
|
43
|
+
" set -a; . ./.env.acceptance; set +a"
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def chosen_model(model: str | None = None) -> str:
|
|
48
|
+
"""The model a session will actually run on.
|
|
49
|
+
|
|
50
|
+
Passed to the session explicitly as well as through the environment. The environment alone
|
|
51
|
+
does not win: the CLI has its own default and will quietly use it, so a run meant for Haiku
|
|
52
|
+
goes out on whatever the CLI felt like and the bill says so afterwards.
|
|
53
|
+
|
|
54
|
+
With nothing named anywhere, the selected backend's own default runs, so switching
|
|
55
|
+
``ALK_HARNESS`` never sends one vendor's model name to another vendor's loop.
|
|
56
|
+
"""
|
|
57
|
+
return (
|
|
58
|
+
model
|
|
59
|
+
or os.environ.get("ALK_HARNESS_MODEL")
|
|
60
|
+
or resolve().default_model
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def thinking_config() -> dict[str, Any]:
|
|
65
|
+
"""How much the model may think, from ALK_HARNESS_THINKING.
|
|
66
|
+
|
|
67
|
+
The Claude Code CLI defaults to adaptive thinking. In this harness the correctness of what a
|
|
68
|
+
stage produces is re-checked by code gates (a scenario is proved against the real world, a
|
|
69
|
+
contract is validated), so the model's private reasoning is spent on decisions the gates make
|
|
70
|
+
again anyway. Left unset, that reasoning was the majority of generated tokens and the majority
|
|
71
|
+
of wall time. Default to disabled for speed; ``adaptive`` restores the old behaviour, and an
|
|
72
|
+
integer sets an explicit budget for models that still honour one.
|
|
73
|
+
"""
|
|
74
|
+
setting = os.environ.get("ALK_HARNESS_THINKING", "disabled").strip().lower()
|
|
75
|
+
if setting in {"adaptive", "on", "auto"}:
|
|
76
|
+
return {"type": "adaptive", "display": "omitted"}
|
|
77
|
+
if setting.isdigit() and int(setting) > 0:
|
|
78
|
+
return {"type": "enabled", "budget_tokens": int(setting), "display": "omitted"}
|
|
79
|
+
return {"type": "disabled"}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def provisioning(enabled: bool | None = None) -> bool:
|
|
83
|
+
"""Compatibility switch for callers selecting the legacy provisioning surface.
|
|
84
|
+
|
|
85
|
+
The autonomous workflow now discovers and provisions source infrastructure automatically;
|
|
86
|
+
explicit stage consumers can still select the older engine-provisioning tool surface while
|
|
87
|
+
they migrate.
|
|
88
|
+
"""
|
|
89
|
+
if enabled is not None:
|
|
90
|
+
return enabled
|
|
91
|
+
return os.environ.get("ALK_HARNESS_PROVISION", "").strip().lower() in {
|
|
92
|
+
"1",
|
|
93
|
+
"true",
|
|
94
|
+
"yes",
|
|
95
|
+
"on",
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def provider_env(model: str | None = None) -> dict[str, str]:
|
|
100
|
+
"""The provider block passed to the session.
|
|
101
|
+
|
|
102
|
+
Claude Code resolves the GCP project from ``GOOGLE_CLOUD_PROJECT``, the credential file, or
|
|
103
|
+
the active gcloud configuration, in that order, so an unset project id is not an error here.
|
|
104
|
+
"""
|
|
105
|
+
# Every model a session can reach is pinned to the same one. Naming only the main model
|
|
106
|
+
# leaves the sub-agent and fast-path settings to the CLI's own preference, and a suite written
|
|
107
|
+
# by twenty writers then runs on whatever that preference happens to be rather than on the
|
|
108
|
+
# model the run asked for.
|
|
109
|
+
chosen = chosen_model(model)
|
|
110
|
+
env = {
|
|
111
|
+
"CLAUDE_CODE_USE_VERTEX": "1",
|
|
112
|
+
"CLOUD_ML_REGION": os.environ.get("CLOUD_ML_REGION", "global"),
|
|
113
|
+
"ANTHROPIC_MODEL": chosen,
|
|
114
|
+
"ANTHROPIC_DEFAULT_SONNET_MODEL": chosen,
|
|
115
|
+
"ANTHROPIC_DEFAULT_OPUS_MODEL": chosen,
|
|
116
|
+
"ANTHROPIC_DEFAULT_HAIKU_MODEL": chosen,
|
|
117
|
+
"ANTHROPIC_SMALL_FAST_MODEL": chosen,
|
|
118
|
+
"CLAUDE_CODE_SUBAGENT_MODEL": chosen,
|
|
119
|
+
}
|
|
120
|
+
for passthrough in (
|
|
121
|
+
"ANTHROPIC_VERTEX_PROJECT_ID",
|
|
122
|
+
"GOOGLE_CLOUD_PROJECT",
|
|
123
|
+
"GOOGLE_APPLICATION_CREDENTIALS",
|
|
124
|
+
):
|
|
125
|
+
value = os.environ.get(passthrough)
|
|
126
|
+
if value:
|
|
127
|
+
env[passthrough] = value
|
|
128
|
+
return env
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def read_only_session(
|
|
132
|
+
*,
|
|
133
|
+
system_prompt: str,
|
|
134
|
+
cwd: str | Path,
|
|
135
|
+
servers: dict[str, Any] | None = None,
|
|
136
|
+
extra_builtins: Iterable[str] = (),
|
|
137
|
+
max_turns: int = 40,
|
|
138
|
+
model: str | None = None,
|
|
139
|
+
) -> SessionSpec:
|
|
140
|
+
"""A session that may read the agent under test but never write to it.
|
|
141
|
+
|
|
142
|
+
The agent under test is somebody's real repository. The harness reads it and writes its own
|
|
143
|
+
artifacts elsewhere, so the built-in write tools are simply not granted; the only way this
|
|
144
|
+
session can produce anything is by calling one of ours.
|
|
145
|
+
"""
|
|
146
|
+
return SessionSpec(
|
|
147
|
+
system_prompt=system_prompt,
|
|
148
|
+
servers=dict(servers or {}),
|
|
149
|
+
builtins=tuple(
|
|
150
|
+
dict.fromkeys([*_READ_ONLY_TOOLS, "AskUserQuestion", *extra_builtins])
|
|
151
|
+
),
|
|
152
|
+
cwd=str(cwd),
|
|
153
|
+
max_turns=max_turns,
|
|
154
|
+
model=chosen_model(model),
|
|
155
|
+
thinking=True,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# Tools the host offers every session that no stage of this harness has any use for. Denying
|
|
160
|
+
# them at the gate works and is the backstop, but a denial still costs the turn that discovered
|
|
161
|
+
# it — and these get reached for in almost every stage. Naming them as disallowed keeps them out
|
|
162
|
+
# of the tool list the model is shown, so the turn is never spent.
|
|
163
|
+
UNWANTED = (
|
|
164
|
+
"ToolSearch",
|
|
165
|
+
"Bash",
|
|
166
|
+
"Write",
|
|
167
|
+
"Edit",
|
|
168
|
+
"NotebookEdit",
|
|
169
|
+
"WebFetch",
|
|
170
|
+
"WebSearch",
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def gate_hooks(granted: Iterable[str]) -> dict[str, Any]:
|
|
175
|
+
"""Deny anything a stage was not given, at the point the SDK actually asks.
|
|
176
|
+
|
|
177
|
+
``can_use_tool`` alone does not do this. An ``allowed_tools`` entry approves those tools
|
|
178
|
+
before the callback is consulted, and the SDK then warns that the callback is shadowed — so
|
|
179
|
+
the gate never runs for the tools we granted, and in practice does not stop the ones we did
|
|
180
|
+
not either. A host ``ToolSearch`` reached every stage, returned nothing, and cost a turn each
|
|
181
|
+
time.
|
|
182
|
+
|
|
183
|
+
A PreToolUse hook is consulted for every call, which is what the deny-by-default rule needed
|
|
184
|
+
in order to be true rather than intended.
|
|
185
|
+
"""
|
|
186
|
+
from claude_agent_sdk.types import HookMatcher
|
|
187
|
+
|
|
188
|
+
permitted = {*granted, "AskUserQuestion"}
|
|
189
|
+
|
|
190
|
+
async def refuse(
|
|
191
|
+
payload: dict[str, Any], _tool_use_id: Any, _context: Any
|
|
192
|
+
) -> dict[str, Any]:
|
|
193
|
+
name = str(payload.get("tool_name") or "")
|
|
194
|
+
if not name or name in permitted:
|
|
195
|
+
return {}
|
|
196
|
+
return {
|
|
197
|
+
"hookSpecificOutput": {
|
|
198
|
+
"hookEventName": "PreToolUse",
|
|
199
|
+
"permissionDecision": "deny",
|
|
200
|
+
"permissionDecisionReason": (
|
|
201
|
+
f"{name} is not part of this stage. You have "
|
|
202
|
+
f"{', '.join(sorted(permitted)) or 'no other tools'}, and everything you "
|
|
203
|
+
"produce goes through those, because those are what check it."
|
|
204
|
+
),
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
return {"PreToolUse": [HookMatcher(hooks=[refuse])]}
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def permission_gate(ask: Any | None = None, granted: Iterable[str] = ()) -> Any:
|
|
212
|
+
"""Decide what a stage may do: nothing it was not given.
|
|
213
|
+
|
|
214
|
+
Deny by default, not deny-a-list. A session is offered whatever tools its host happens to
|
|
215
|
+
expose, and anything not named here is by definition not part of how this stage works. An
|
|
216
|
+
allow-by-default gate let a host search tool through, which returned nothing useful and cost
|
|
217
|
+
a stage its entire turn budget looping on it; the same hole would let a file write through.
|
|
218
|
+
|
|
219
|
+
Tools granted through ``allowed_tools`` are approved before this is consulted, so this only
|
|
220
|
+
ever sees the ones that were not.
|
|
221
|
+
"""
|
|
222
|
+
permitted = set(granted)
|
|
223
|
+
|
|
224
|
+
async def gate(tool_name: str, payload: dict[str, Any], context: Any) -> Any:
|
|
225
|
+
from claude_agent_sdk.types import PermissionResultAllow, PermissionResultDeny
|
|
226
|
+
|
|
227
|
+
if tool_name == "AskUserQuestion" and ask is not None:
|
|
228
|
+
return await ask(tool_name, payload, context)
|
|
229
|
+
if tool_name in permitted:
|
|
230
|
+
return PermissionResultAllow(updated_input=payload)
|
|
231
|
+
return PermissionResultDeny(
|
|
232
|
+
message=(
|
|
233
|
+
f"{tool_name} is not part of this stage. You have "
|
|
234
|
+
f"{', '.join(sorted(permitted)) or 'no other tools'}, and everything you "
|
|
235
|
+
"produce goes through those, because those are what check it."
|
|
236
|
+
)
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
return gate
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def artifact_dir(agent: str, root: str | Path | None = None) -> Path:
|
|
243
|
+
"""The folder holding one conversation: its contract, world, scenarios and runs.
|
|
244
|
+
|
|
245
|
+
One conversation, one directory. Everything about testing one agent lives together, which is
|
|
246
|
+
what makes a session something you can close, reopen, hand over or delete as one thing.
|
|
247
|
+
"""
|
|
248
|
+
base = Path(root) if root else ARTIFACTS_ROOT / "sessions"
|
|
249
|
+
return base / agent
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
HARNESS = SKILLS_ROOT / "harness.md"
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def discovered_skills(**about: str) -> str:
|
|
256
|
+
"""Every extra skill that says it applies to this agent, found by looking rather than by name.
|
|
257
|
+
|
|
258
|
+
A skill is a markdown file under ``skills/kinds/`` whose first lines declare what it is for::
|
|
259
|
+
|
|
260
|
+
---
|
|
261
|
+
name: voice
|
|
262
|
+
applies_to: modality=voice
|
|
263
|
+
---
|
|
264
|
+
|
|
265
|
+
The harness reads the directory, keeps the files whose ``applies_to`` matches what it was told
|
|
266
|
+
about this agent, and appends them in name order. ``applies_to: any`` always matches, and a file
|
|
267
|
+
with no declaration is skipped rather than guessed at.
|
|
268
|
+
|
|
269
|
+
Naming each kind in code would mean editing code to add one, and there will be many: voice, chat,
|
|
270
|
+
browser, and whatever a customer turns up with next. **Adding support for a kind of agent is
|
|
271
|
+
adding a file here.**
|
|
272
|
+
"""
|
|
273
|
+
root = SKILLS_ROOT / "kinds"
|
|
274
|
+
if not root.is_dir():
|
|
275
|
+
return ""
|
|
276
|
+
wanted = {
|
|
277
|
+
key.lower(): str(value).strip().lower() for key, value in about.items() if value
|
|
278
|
+
}
|
|
279
|
+
found: list[tuple[str, str]] = []
|
|
280
|
+
for path in sorted(root.glob("*.md")):
|
|
281
|
+
text = path.read_text(encoding="utf-8")
|
|
282
|
+
head = text.split("---")[1] if text.startswith("---") and "---" in text[3:] else ""
|
|
283
|
+
applies = ""
|
|
284
|
+
for line in head.splitlines():
|
|
285
|
+
if line.strip().lower().startswith("applies_to:"):
|
|
286
|
+
applies = line.split(":", 1)[1].strip().lower()
|
|
287
|
+
if not applies:
|
|
288
|
+
continue
|
|
289
|
+
if applies == "any":
|
|
290
|
+
found.append((path.stem, text))
|
|
291
|
+
continue
|
|
292
|
+
# `key=value`, and every clause has to hold.
|
|
293
|
+
clauses = [one.strip() for one in applies.split(",") if one.strip()]
|
|
294
|
+
if all(
|
|
295
|
+
"=" in clause
|
|
296
|
+
and wanted.get(clause.split("=", 1)[0].strip())
|
|
297
|
+
== clause.split("=", 1)[1].strip()
|
|
298
|
+
for clause in clauses
|
|
299
|
+
):
|
|
300
|
+
found.append((path.stem, text))
|
|
301
|
+
if not found:
|
|
302
|
+
return ""
|
|
303
|
+
return "\n\n---\n\n" + "\n\n---\n\n".join(text for _name, text in found)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def load_skill(name: str) -> str:
|
|
307
|
+
"""One stage's instructions, behind what the harness as a whole is for.
|
|
308
|
+
|
|
309
|
+
Every stage gets the same opening: what this harness produces, why the division between what
|
|
310
|
+
a model decides and what code decides exists, and what makes a result worth believing. A
|
|
311
|
+
stage that knows only its own step does its step well and still gets the point of it wrong —
|
|
312
|
+
it works around a gate instead of fixing what the gate named, or it reports a number that
|
|
313
|
+
quietly skipped half its checks.
|
|
314
|
+
|
|
315
|
+
The stage's own method follows. Both are files, so how any of this works can be changed
|
|
316
|
+
without touching code.
|
|
317
|
+
"""
|
|
318
|
+
path = SKILLS_ROOT / name / "SKILL.md"
|
|
319
|
+
if not path.exists():
|
|
320
|
+
raise FileNotFoundError(f"no skill at {path}")
|
|
321
|
+
stage = path.read_text(encoding="utf-8")
|
|
322
|
+
# Lookup material a skill keeps beside itself, carried in with it. A session runs with its working
|
|
323
|
+
# directory on the artifacts it is producing, not on the skills tree, and the skills live inside an
|
|
324
|
+
# installed package, so a skill that says "see references/x.md" is naming a path its reader cannot
|
|
325
|
+
# reach. Appending them is what makes the split into a main file and its references safe.
|
|
326
|
+
for reference in sorted((SKILLS_ROOT / name / "references").glob("*.md")):
|
|
327
|
+
stage += (
|
|
328
|
+
f"\n\n---\n\n# references/{reference.name}\n\n"
|
|
329
|
+
f"{reference.read_text(encoding='utf-8')}"
|
|
330
|
+
)
|
|
331
|
+
if not HARNESS.exists():
|
|
332
|
+
return stage
|
|
333
|
+
return (
|
|
334
|
+
f"{HARNESS.read_text(encoding='utf-8')}\n\n"
|
|
335
|
+
"---\n\n"
|
|
336
|
+
"# The stage you are in now\n\n"
|
|
337
|
+
f"{stage}"
|
|
338
|
+
)
|