agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/harness/README.md
ADDED
|
@@ -0,0 +1,417 @@
|
|
|
1
|
+
# The harness
|
|
2
|
+
|
|
3
|
+
Point it at an agent. It reads the agent, builds a real database its tools run against, writes
|
|
4
|
+
test scenarios, runs them as conversations, and tells you what held and what did not.
|
|
5
|
+
|
|
6
|
+
Nothing here is written for a particular agent. Every stage takes the contract and the world as
|
|
7
|
+
input, so a different agent is the same commands with a different name.
|
|
8
|
+
|
|
9
|
+
The normal product path is autonomous—no operator messages or stage-by-stage nudges:
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
agent-learn harness auto \
|
|
13
|
+
--path /absolute/path/to/private-agent \
|
|
14
|
+
--count 10
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
This creates one session containing `job.json`, a sealed `environment-bundle/`, contract,
|
|
18
|
+
generated world/data, validated scenarios, canonical progress events, calls, evidence and run
|
|
19
|
+
artifacts. Agent check failures still complete the run and remain visible as RL evidence.
|
|
20
|
+
|
|
21
|
+
The same `HarnessJob` and `HarnessExecutor` run in Future AGI's isolated hosted sandbox. The
|
|
22
|
+
platform creates jobs and stores their events/artifacts; it does not execute harness stages. See
|
|
23
|
+
[ARCHITECTURE.md](ARCHITECTURE.md) for boundaries, isolation and extension points.
|
|
24
|
+
|
|
25
|
+
Repository-backed chat agents now follow the same runtime lifecycle as voice agents when their
|
|
26
|
+
submitted service exposes an HTTP (ALK or OpenAI-compatible) or JSON WebSocket ingress. ALK starts
|
|
27
|
+
the real service per scenario, connects the simulated user, routes declared environment endpoints,
|
|
28
|
+
records tool/state evidence and tears the service down. It does not reconstruct a repository agent
|
|
29
|
+
from its prompt when no conversational ingress exists.
|
|
30
|
+
|
|
31
|
+
---
|
|
32
|
+
|
|
33
|
+
# Part 1 — Setting up, from nothing
|
|
34
|
+
|
|
35
|
+
If you have never run this before, do these five steps in order. They take about ten minutes,
|
|
36
|
+
most of which is waiting for the install.
|
|
37
|
+
|
|
38
|
+
## Before you start
|
|
39
|
+
|
|
40
|
+
You need four things on your machine:
|
|
41
|
+
|
|
42
|
+
| What | Check it with | If missing |
|
|
43
|
+
|---|---|---|
|
|
44
|
+
| Python 3.10 or newer | `python3 --version` | install from python.org, or `brew install python` |
|
|
45
|
+
| `uv` (the package manager this repo uses) | `uv --version` | `brew install uv` |
|
|
46
|
+
| The `claude` command | `claude --version` | `npm install -g @anthropic-ai/claude-code` |
|
|
47
|
+
| A Google Cloud service-account key file (`.json`) for Vertex AI | you were given one, or ask | ask whoever set up your GCP access |
|
|
48
|
+
|
|
49
|
+
The `claude` command matters: the harness talks to the model through the Claude Agent SDK, and
|
|
50
|
+
that SDK runs the `claude` binary under the hood. If it is not installed, every stage fails
|
|
51
|
+
immediately with a connection error.
|
|
52
|
+
|
|
53
|
+
## Step 1. Get the repo, and work from its root
|
|
54
|
+
|
|
55
|
+
Every command in this document is run from the **root of the repo**, not from this folder:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
git clone https://github.com/future-agi/agent-learning-kit
|
|
59
|
+
cd agent-learning-kit
|
|
60
|
+
git checkout feat/environment-generation # until this branch is merged
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Wherever you cloned it, that directory is the one containing `pyproject.toml`. Check you are in
|
|
64
|
+
the right place:
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
ls pyproject.toml # should print: pyproject.toml
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
If that errors, you are in the wrong directory. Do not continue until it works.
|
|
71
|
+
|
|
72
|
+
## Step 2 — Install the dependencies
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
uv sync --extra livekit --group dev
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
This reads `pyproject.toml`, downloads everything, and creates a folder called `.venv` in the
|
|
79
|
+
repo root. That folder is the "virtual environment": a private copy of Python with this
|
|
80
|
+
project's packages in it, so they do not collide with anything else on your machine.
|
|
81
|
+
|
|
82
|
+
The `--extra livekit` matters even though the harness never makes a voice call. The harness
|
|
83
|
+
builds on `fi.simulate.environment`, and importing anything from `fi.simulate` runs that
|
|
84
|
+
package's `__init__`, which pulls in its LiveKit scenario generator. Plain `uv sync` leaves that
|
|
85
|
+
out and every command dies with `No module named 'livekit'`.
|
|
86
|
+
|
|
87
|
+
It takes a few minutes the first time. You only do this once.
|
|
88
|
+
|
|
89
|
+
## Step 3 — Use the virtual environment
|
|
90
|
+
|
|
91
|
+
Two ways. **Pick one and stick with it.**
|
|
92
|
+
|
|
93
|
+
**Option A — no activation (what this document uses).** Call the Python inside `.venv` directly:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
.venv/bin/python -m fi.alk.harness
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Nothing to remember, nothing to undo, works in a fresh terminal every time. Every command below
|
|
100
|
+
is written this way.
|
|
101
|
+
|
|
102
|
+
**Option B — activate it.** If you prefer typing plain `python`:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
source .venv/bin/activate # your prompt now shows (agent-learning-kit)
|
|
106
|
+
python -m fi.alk.harness # plain "python" now means the one in .venv
|
|
107
|
+
deactivate # when you are done
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Activation only lasts for that terminal window. Open a new tab and you must activate again. If a
|
|
111
|
+
command ever fails with `No module named fi`, you almost certainly forgot.
|
|
112
|
+
|
|
113
|
+
## Step 4 — Credentials
|
|
114
|
+
|
|
115
|
+
The harness reaches the model through Vertex AI, which needs your Google Cloud service-account
|
|
116
|
+
key. Nothing is hardcoded and no key is ever read from source.
|
|
117
|
+
|
|
118
|
+
Create a local env file from the template that ships with the repo:
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
cp oss/simulation-acceptance/.env.example .env.acceptance
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Open `.env.acceptance` in an editor and fill in two lines:
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/your-service-account.json
|
|
128
|
+
GOOGLE_CLOUD_PROJECT=your-gcp-project-id
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
`.env.acceptance` is git-ignored. It holds a path to a private key: **never commit it, never
|
|
132
|
+
paste its contents into Slack or a PR.**
|
|
133
|
+
|
|
134
|
+
Now load it into your terminal, and pick a model:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
set -a; . ./.env.acceptance; set +a
|
|
138
|
+
export CLOUD_ML_REGION=global
|
|
139
|
+
export ALK_HARNESS_MODEL=claude-sonnet-4-6
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
- `set -a; . ./file; set +a` means "read this file and export everything in it". The leading
|
|
143
|
+
`. ` (dot space) is what runs it in your *current* shell, so the variables stick around.
|
|
144
|
+
- `ALK_HARNESS_MODEL` picks the model. **Use `claude-sonnet-4-6` or better.** Haiku is cheaper
|
|
145
|
+
but has twice misread an agent's modality, and modality decides how every later test is run.
|
|
146
|
+
|
|
147
|
+
These last only for the current terminal window. Every new terminal, run these three lines again.
|
|
148
|
+
|
|
149
|
+
## Step 5 — Check it works
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
.venv/bin/python -m pytest tests/test_harness.py -q
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
These are offline tests: no model calls, no credentials, no network. If they pass, your
|
|
156
|
+
install is fine. If they fail, the problem is Step 2, not your credentials.
|
|
157
|
+
|
|
158
|
+
Then check the credentials separately, with the cheapest thing that talks to the model:
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
.venv/bin/python -m fi.alk.harness
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Say hello. If it answers, the credentials work; type `q` to leave before it spends anything
|
|
165
|
+
real.
|
|
166
|
+
|
|
167
|
+
---
|
|
168
|
+
|
|
169
|
+
# Part 2 — Using it
|
|
170
|
+
|
|
171
|
+
## The short version
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
cd path/to/agent-learning-kit
|
|
175
|
+
set -a; . ./.env.acceptance; set +a
|
|
176
|
+
export CLOUD_ML_REGION=global ALK_HARNESS_MODEL=claude-sonnet-4-6
|
|
177
|
+
|
|
178
|
+
.venv/bin/python harness-ui/server.py # a web page, on :8777
|
|
179
|
+
.venv/bin/python -m fi.alk.harness # the same thing in the terminal
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Either one is the whole interface. Both open with "which agent would you like to test, and where
|
|
183
|
+
is it?", and everything after that is a conversation. It finds the agent, reads it, builds the
|
|
184
|
+
world, writes the scenarios, and runs them, moving on as each stage produces its artifact.
|
|
185
|
+
|
|
186
|
+
**The page is the one to start with**: it shows what each stage produced while you talk, and it
|
|
187
|
+
is the same harness underneath. There is nothing separate to build or serve; see
|
|
188
|
+
`harness-ui/README.md`.
|
|
189
|
+
|
|
190
|
+
One message is enough to begin:
|
|
191
|
+
|
|
192
|
+
```
|
|
193
|
+
i want to test my voice ordering agent. the code is at /absolute/path/to/the/agent
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
In the terminal version: type what you want and press enter, press enter on an **empty** line to
|
|
197
|
+
move to the next stage, and type `q` to leave.
|
|
198
|
+
|
|
199
|
+
## Where things are written
|
|
200
|
+
|
|
201
|
+
One conversation, one folder. Everything about testing one agent lives together, so closing the
|
|
202
|
+
page, restarting the server or coming back tomorrow all resume by reading the folder.
|
|
203
|
+
|
|
204
|
+
```
|
|
205
|
+
artifacts/sessions/<id>/
|
|
206
|
+
session.json which agent, where its source is, when it started
|
|
207
|
+
chat.jsonl the conversation itself
|
|
208
|
+
contract.json what the agent verifiably is
|
|
209
|
+
world.sqlite the world, with handlers/, simulator_prompt.md, sub_goals.json
|
|
210
|
+
scenarios/<name>/ one folder per scenario
|
|
211
|
+
runs.json what happened when they ran
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
The id is readable and unique (`drive-thru-aaea25`), so two attempts at the same agent are two
|
|
215
|
+
sessions rather than one overwriting the other. To start from nothing:
|
|
216
|
+
`rm -rf artifacts/sessions/* artifacts/.open-session`.
|
|
217
|
+
|
|
218
|
+
## The same stages, one at a time
|
|
219
|
+
|
|
220
|
+
Useful when you want to redo one thing without walking the whole conversation. Each of these
|
|
221
|
+
stays open for corrections until you type `q`; add `--once` to run it unattended and exit.
|
|
222
|
+
|
|
223
|
+
```bash
|
|
224
|
+
# read an agent's source and write down what it verifiably is
|
|
225
|
+
.venv/bin/python -m fi.alk.harness understand --name my_agent --path ../my-agent-repo
|
|
226
|
+
|
|
227
|
+
# build the environment: the world, the simulator prompt, the sub-goal catalogue
|
|
228
|
+
.venv/bin/python -m fi.alk.harness build --name my_agent
|
|
229
|
+
|
|
230
|
+
# write the test scenarios, each proved before it is kept
|
|
231
|
+
.venv/bin/python -m fi.alk.harness scenarios --name my_agent --count 10
|
|
232
|
+
|
|
233
|
+
# run them against the world here, and grade
|
|
234
|
+
.venv/bin/python -m fi.alk.harness run --name my_agent
|
|
235
|
+
|
|
236
|
+
# or run them against the real hosted agent, as a conversation
|
|
237
|
+
.venv/bin/python -m fi.alk.harness live --name my_agent
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
`--name` is just a label for the folder your artifacts go in. `--path` is where the agent's code
|
|
241
|
+
lives — a path to another repo on your disk.
|
|
242
|
+
|
|
243
|
+
Useful extras:
|
|
244
|
+
|
|
245
|
+
- `run --only <name> [<name> ...]` runs a single scenario instead of all of them
|
|
246
|
+
- `run --quiet` hides the conversation and prints only verdicts
|
|
247
|
+
- `scenarios` without `--count` uses however many already exist, because coming back to change
|
|
248
|
+
one is not a request for a different number of them
|
|
249
|
+
|
|
250
|
+
## What each stage does
|
|
251
|
+
|
|
252
|
+
**understand** reads the agent's source and produces `contract.json`: its tools, the exact
|
|
253
|
+
argument names and permitted values, its hard rules, its real data. Everything downstream is
|
|
254
|
+
confined to this, which is what stops later stages inventing tools or menu items. Anything
|
|
255
|
+
changed later goes through an amendment tool and is recorded with its reason, so what came from
|
|
256
|
+
the agent and what came from us stay distinguishable.
|
|
257
|
+
|
|
258
|
+
**build** produces everything common to every test of this agent:
|
|
259
|
+
|
|
260
|
+
- **the world** — a real database behind the agent's tools, with one handler per tool that can
|
|
261
|
+
genuinely refuse: a nonexistent id, an unavailable item, an argument outside what the tool
|
|
262
|
+
accepts. A refusal is the world working; a crash is a defect, and the two are never confused.
|
|
263
|
+
- **the simulator prompt** — for a conversational agent, the person on the other side, written
|
|
264
|
+
once with `{{ slot }}` variables each scenario fills.
|
|
265
|
+
- **the sub-goal catalogue** — the named things this agent can be checked on, each carrying its
|
|
266
|
+
check **as code** wherever the answer is observable, and marked judged only where nothing is.
|
|
267
|
+
|
|
268
|
+
It is exercised before it can be saved — every tool probed with a valid call, a bogus id and a
|
|
269
|
+
missing argument, plus declared sequences where state must carry across calls — and `save_world`
|
|
270
|
+
refuses a world that fails, has no sequences, no sub-goals, only judged sub-goals, no simulator
|
|
271
|
+
prompt for a conversational agent, or rows left over from its own testing.
|
|
272
|
+
|
|
273
|
+
**scenarios** writes each test as a change on that base. Each one owns a folder, and the code in
|
|
274
|
+
it is code, not strings inside a JSON file:
|
|
275
|
+
|
|
276
|
+
```
|
|
277
|
+
scenarios/<name>/
|
|
278
|
+
scenario.json the instruction, the reference solution, which sub-goals it names
|
|
279
|
+
setup.py def setup(world) what this scenario changes first
|
|
280
|
+
ready.py def ready(world) is the world ready for it
|
|
281
|
+
checks/<goal>.py def check(world, calls) one per deterministic sub-goal
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
`setup` is code rather than a list of rows because "not necessarily the database alone" cannot be
|
|
285
|
+
written as rows. The check files genuinely run on their own:
|
|
286
|
+
|
|
287
|
+
```bash
|
|
288
|
+
python scenarios/<name>/checks/<goal>.py path/to/world.sqlite # prints held, or FAILED: ...
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
Before a scenario is kept it is **proved** by three gates, all pure code, no model involved:
|
|
292
|
+
|
|
293
|
+
1. **ready**: reset → `setup` → `ready`. The world must hold what the scenario presumes. A
|
|
294
|
+
scenario about the last five items is only a test of the agent if there really are five;
|
|
295
|
+
otherwise the agent fails for something we got wrong and it reads as the agent's fault.
|
|
296
|
+
2. **solvable**: then run the reference solution and the checks. They must **pass**, or either
|
|
297
|
+
the scenario cannot be passed or a check is wrong.
|
|
298
|
+
3. **not vacuous**: then reset, set up again, run **nothing**, and run the checks. They must
|
|
299
|
+
**fail**. A check that passes while the agent does nothing grades nothing while reporting a
|
|
300
|
+
result.
|
|
301
|
+
|
|
302
|
+
Only a scenario clearing all three is kept. The reference solution is kept with it, and is never
|
|
303
|
+
run against the agent under test.
|
|
304
|
+
|
|
305
|
+
**run** gives each scenario its own restored copy of the world and grades from what is left
|
|
306
|
+
behind: the state of the world plus every tool call with its arguments. `run` converses with the
|
|
307
|
+
agent locally, rebuilt from its contract. `live` is the same grading against the **real hosted
|
|
308
|
+
agent**: the webhook its own tools call is answered by the world, so a call for something that
|
|
309
|
+
is not there is refused rather than mocked into success.
|
|
310
|
+
|
|
311
|
+
## How it grades
|
|
312
|
+
|
|
313
|
+
Deterministic by default, a judge only as the fallback.
|
|
314
|
+
|
|
315
|
+
Every sub-goal with a check in code is settled by running that check against two things the run
|
|
316
|
+
left behind: the world afterwards, and the recorded tool calls with their arguments — so "booked
|
|
317
|
+
10 PM when 11 PM was asked" is caught without any judgement. Sub-goals marked judged are handed
|
|
318
|
+
to a model with three kinds of evidence: what was said, what the agent actually did, and the
|
|
319
|
+
state afterwards. An unanswered claim counts as failed, never as passed, and judged results are
|
|
320
|
+
always reported as judged rather than blended into the code-settled score.
|
|
321
|
+
|
|
322
|
+
```
|
|
323
|
+
PASS quantity_and_unavailable 3/3 sub-goals settled by code
|
|
324
|
+
[x] quantity_honored
|
|
325
|
+
[x] unavailable_drink_refused
|
|
326
|
+
[x] regular_item_placed_correctly
|
|
327
|
+
[?] no_unrequested_items — judged, not settled by code
|
|
328
|
+
|
|
329
|
+
what the agent actually did:
|
|
330
|
+
order_regular_item({'item_id': 'hamburger'}) -> ok
|
|
331
|
+
order_regular_item({'item_id': 'hamburger'}) -> ok
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
A run where the world crashed is `VOID`, not `FAIL` — that says nothing about the agent. A check
|
|
335
|
+
that raises is a **broken check**, reported as ours, never scored against the agent.
|
|
336
|
+
|
|
337
|
+
## What it refuses to do
|
|
338
|
+
|
|
339
|
+
These are the parts worth understanding, because they are what make a result mean something.
|
|
340
|
+
|
|
341
|
+
- A world that fails its own probes will not save; nor will one with no sequences, no sub-goals,
|
|
342
|
+
only judged sub-goals, or rows left over from building it.
|
|
343
|
+
- A scenario is not kept until the world is ready for it, its own solution passes its own checks,
|
|
344
|
+
and those checks fail when nothing is done. Missing preconditions, unsolvable scenarios and
|
|
345
|
+
vacuous checks all die here, at write time.
|
|
346
|
+
- A scenario naming a sub-goal nobody defined, or a table nobody built, is rejected and told
|
|
347
|
+
what does exist.
|
|
348
|
+
- A suite where no sub-goal is shared between scenarios will not save, because nothing would
|
|
349
|
+
roll up across it.
|
|
350
|
+
- Changing the contract is allowed but never silent: every widening, added rule or corrected
|
|
351
|
+
tool is recorded with its reason in `amendments[]`.
|
|
352
|
+
|
|
353
|
+
If a stage tells you it will not do something, that is the design, not a bug to route around.
|
|
354
|
+
|
|
355
|
+
## What a full pass costs, and how long it takes
|
|
356
|
+
|
|
357
|
+
Measured on Sonnet, on a five-tool voice agent, all three stages in one conversation:
|
|
358
|
+
|
|
359
|
+
| Stage | Turns | Time | Cost |
|
|
360
|
+
|---|---|---|---|
|
|
361
|
+
| reading the agent | 6 | under a minute | ~$0.55 |
|
|
362
|
+
| building the environment | 33 | ~10 minutes | ~$1.40 |
|
|
363
|
+
| five proved scenarios | 22 | ~5 minutes | ~$0.92 |
|
|
364
|
+
|
|
365
|
+
About **$3.30 and twenty minutes** end to end. Building the environment is the long stage, and
|
|
366
|
+
**the Environment tab stays empty until it finishes**: the world is held in memory until
|
|
367
|
+
`save_world` writes it. Watch the chat for progress instead. Grading a local run afterwards is a
|
|
368
|
+
few cents per scenario.
|
|
369
|
+
|
|
370
|
+
## When something goes wrong
|
|
371
|
+
|
|
372
|
+
| What you see | What it means |
|
|
373
|
+
|---|---|
|
|
374
|
+
| `No module named fi` | Wrong directory, or you are using system `python` instead of `.venv/bin/python` |
|
|
375
|
+
| `command not found: uv` | `brew install uv` |
|
|
376
|
+
| `No module named 'livekit'` | You ran plain `uv sync`. Run `uv sync --extra livekit --group dev` |
|
|
377
|
+
| `No module named 'fastapi'` | Same cause. The UI's dependencies come in with `--group dev` (or `--extra harness-ui`) |
|
|
378
|
+
| Fails instantly on any model call | The `claude` command is not installed, or your env vars are not loaded in this terminal |
|
|
379
|
+
| `Could not load the default credentials` | `GOOGLE_APPLICATION_CREDENTIALS` is unset or points at a file that is not there |
|
|
380
|
+
| `nobody has said which agent this is about yet` | Say where the agent's code lives, with an absolute path |
|
|
381
|
+
| `No contract at ...` | Read the agent first |
|
|
382
|
+
| `No world at ...` | Build the environment first |
|
|
383
|
+
| The page shows empty tabs | Look at which session is open. A build in progress has not written its world yet |
|
|
384
|
+
| A stage does nothing and exits | It ran out of turns. Look at the last few lines: it usually says what it was stuck on |
|
|
385
|
+
| A change to the harness seems to have no effect | Restart the server. A long-lived process does not reload code or skills |
|
|
386
|
+
| `lsof -ti:8777` says the server is up after you stopped it | That matches a browser's leftover sockets. Use `lsof -nP -iTCP:8777 -sTCP:LISTEN` |
|
|
387
|
+
|
|
388
|
+
Everything a stage did is printed as it happens, and every run is kept in
|
|
389
|
+
`artifacts/sessions/<id>/runs.json`, including the transcript and every tool call.
|
|
390
|
+
|
|
391
|
+
---
|
|
392
|
+
|
|
393
|
+
# Part 3 — For developers
|
|
394
|
+
|
|
395
|
+
## Adding to it
|
|
396
|
+
|
|
397
|
+
- A new **agent** is nothing: the same stages read its contract.
|
|
398
|
+
- A new **kind of world** is a class and a registration in `world/kinds.py`. Browser is registered
|
|
399
|
+
and stubbed; sqlite is the one built out.
|
|
400
|
+
- A new **place the agent runs** is a class and a registration in `run/targets.py`. `local` runs
|
|
401
|
+
the agent here from its contract; the live voice path answers a hosted assistant's webhook from
|
|
402
|
+
the same `world.handle_tool_call`, so the world, the scenarios and the grading do not change.
|
|
403
|
+
- A change to **how a stage works** is an edit to its `skills/<stage>/SKILL.md`. The markdown is
|
|
404
|
+
the method; code holds only what must be exact.
|
|
405
|
+
|
|
406
|
+
## Not done yet
|
|
407
|
+
|
|
408
|
+
- Browser worlds are registered but not built.
|
|
409
|
+
- Snapshots are local files, not object storage.
|
|
410
|
+
- Judged sub-goals on the live path are reported as judged, not yet sent to a judge.
|
|
411
|
+
- Nothing reports which of the contract's use cases have no scenario.
|
|
412
|
+
|
|
413
|
+
## Tests
|
|
414
|
+
|
|
415
|
+
```bash
|
|
416
|
+
.venv/bin/python -m pytest tests/test_harness.py -q # offline, no credentials needed
|
|
417
|
+
```
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""The harness: an agent that builds test environments for other agents.
|
|
2
|
+
|
|
3
|
+
It reads an agent, works out what it verifiably is, builds a world its tools can run against,
|
|
4
|
+
generates scenarios, runs them, and reads the results back. Each of those is a stage, each stage
|
|
5
|
+
is its own session, and stages hand work to each other as artifacts on disk.
|
|
6
|
+
|
|
7
|
+
The split that matters: the model does judgement, and code decides outcomes. Reading unfamiliar
|
|
8
|
+
source, designing a schema, and choosing what is worth testing are judgement. Executing a tool
|
|
9
|
+
call and grading a run are not, and are never delegated to a model.
|
|
10
|
+
|
|
11
|
+
Stages are described in files under ``skills/``, so the method is editable without touching
|
|
12
|
+
code, and where an agent comes from is a registered source, so a new kind of agent is a class
|
|
13
|
+
rather than a new code path.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from .chat import Conversation, open_conversation
|
|
17
|
+
from .bundle import EnvironmentBundle, load_bundle, seal_bundle
|
|
18
|
+
from .environment_plan import EnvironmentPlan, load_environment_plan
|
|
19
|
+
from .config import (
|
|
20
|
+
DEFAULT_MODEL,
|
|
21
|
+
artifact_dir,
|
|
22
|
+
load_skill,
|
|
23
|
+
provider_env,
|
|
24
|
+
read_only_session,
|
|
25
|
+
)
|
|
26
|
+
from .contract import AgentContract, Runtime, RuntimeInterface, ToolSpec, validate_contract
|
|
27
|
+
from .job import ExecutionMode, HarnessJob, HarnessStage
|
|
28
|
+
from .scenario import Scenario, validate_scenario
|
|
29
|
+
from .session import Stage, Turn
|
|
30
|
+
from .sources import (
|
|
31
|
+
AgentSource,
|
|
32
|
+
GitHubSource,
|
|
33
|
+
ProviderSource,
|
|
34
|
+
RepoSource,
|
|
35
|
+
SpecSource,
|
|
36
|
+
register_source,
|
|
37
|
+
resolve,
|
|
38
|
+
supported,
|
|
39
|
+
)
|
|
40
|
+
from .understand import open_stage, understand
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"AgentContract",
|
|
44
|
+
"AgentSource",
|
|
45
|
+
"Conversation",
|
|
46
|
+
"DEFAULT_MODEL",
|
|
47
|
+
"EnvironmentBundle",
|
|
48
|
+
"EnvironmentPlan",
|
|
49
|
+
"ExecutionMode",
|
|
50
|
+
"GitHubSource",
|
|
51
|
+
"HarnessJob",
|
|
52
|
+
"HarnessStage",
|
|
53
|
+
"ProviderSource",
|
|
54
|
+
"RepoSource",
|
|
55
|
+
"Runtime",
|
|
56
|
+
"RuntimeInterface",
|
|
57
|
+
"Scenario",
|
|
58
|
+
"SpecSource",
|
|
59
|
+
"Stage",
|
|
60
|
+
"ToolSpec",
|
|
61
|
+
"Turn",
|
|
62
|
+
"artifact_dir",
|
|
63
|
+
"load_skill",
|
|
64
|
+
"load_bundle",
|
|
65
|
+
"load_environment_plan",
|
|
66
|
+
"open_conversation",
|
|
67
|
+
"open_stage",
|
|
68
|
+
"provider_env",
|
|
69
|
+
"read_only_session",
|
|
70
|
+
"register_source",
|
|
71
|
+
"resolve",
|
|
72
|
+
"seal_bundle",
|
|
73
|
+
"supported",
|
|
74
|
+
"understand",
|
|
75
|
+
"validate_contract",
|
|
76
|
+
"validate_scenario",
|
|
77
|
+
]
|