agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""HTTP bridge embedded into hosted bundles for ``fi.simulate`` callbacks.
|
|
2
|
+
|
|
3
|
+
The bundle producer serializes this module into the process command. It deliberately uses only
|
|
4
|
+
the standard library plus the submitted agent's own ``fi.simulate`` dependency, so a callback-only
|
|
5
|
+
repository does not need to be modified or taught about the hosted process runtime.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import importlib
|
|
12
|
+
import inspect
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
import time
|
|
17
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _load_callback() -> Any:
|
|
22
|
+
target = os.environ["ALK_CALLBACK_ENTRYPOINT"]
|
|
23
|
+
module_name, separator, attribute = target.partition(":")
|
|
24
|
+
if not separator or not module_name or not attribute:
|
|
25
|
+
raise RuntimeError(f"invalid ALK_CALLBACK_ENTRYPOINT: {target!r}")
|
|
26
|
+
callback = getattr(importlib.import_module(module_name), attribute)
|
|
27
|
+
if not callable(callback):
|
|
28
|
+
raise RuntimeError(f"callback entrypoint is not callable: {target!r}")
|
|
29
|
+
return callback
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
CALLBACK = _load_callback()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _json_body(value: Any) -> dict[str, Any]:
|
|
36
|
+
if hasattr(value, "model_dump"):
|
|
37
|
+
body = value.model_dump(mode="json", exclude_none=True)
|
|
38
|
+
elif isinstance(value, dict):
|
|
39
|
+
body = value
|
|
40
|
+
else:
|
|
41
|
+
body = {"content": str(value)}
|
|
42
|
+
if not isinstance(body, dict):
|
|
43
|
+
raise TypeError("callback response must serialize to a JSON object")
|
|
44
|
+
body.setdefault("content", "")
|
|
45
|
+
return body
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _invoke(payload: dict[str, Any]) -> dict[str, Any]:
|
|
49
|
+
from fi.simulate import AgentInput
|
|
50
|
+
|
|
51
|
+
result = CALLBACK(AgentInput.model_validate(payload))
|
|
52
|
+
if inspect.isawaitable(result):
|
|
53
|
+
result = asyncio.run(result)
|
|
54
|
+
return _json_body(result)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class Handler(BaseHTTPRequestHandler):
|
|
58
|
+
def _respond(self, status: int, body: dict[str, Any]) -> None:
|
|
59
|
+
encoded = json.dumps(body, separators=(",", ":")).encode("utf-8")
|
|
60
|
+
self.send_response(status)
|
|
61
|
+
self.send_header("Content-Type", "application/json")
|
|
62
|
+
self.send_header("Content-Length", str(len(encoded)))
|
|
63
|
+
self.end_headers()
|
|
64
|
+
self.wfile.write(encoded)
|
|
65
|
+
|
|
66
|
+
def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
67
|
+
if self.path.rstrip("/") == "/health":
|
|
68
|
+
self._respond(200, {"status": "ok"})
|
|
69
|
+
else:
|
|
70
|
+
self._respond(404, {"error": "not_found"})
|
|
71
|
+
|
|
72
|
+
def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
73
|
+
if self.path.rstrip("/") != "/invoke":
|
|
74
|
+
self._respond(404, {"error": "not_found"})
|
|
75
|
+
return
|
|
76
|
+
started = time.monotonic()
|
|
77
|
+
status = 500
|
|
78
|
+
error_type = "none"
|
|
79
|
+
try:
|
|
80
|
+
length = int(self.headers.get("Content-Length", "0"))
|
|
81
|
+
payload = json.loads(self.rfile.read(length) or b"{}")
|
|
82
|
+
if not isinstance(payload, dict):
|
|
83
|
+
raise TypeError("request body must be a JSON object")
|
|
84
|
+
self._respond(200, _invoke(payload))
|
|
85
|
+
status = 200
|
|
86
|
+
except Exception as exc: # noqa: BLE001 - target errors cross the HTTP seam
|
|
87
|
+
error_type = type(exc).__name__
|
|
88
|
+
self._respond(
|
|
89
|
+
500,
|
|
90
|
+
{"error": f"{type(exc).__name__}: {exc}", "content": ""},
|
|
91
|
+
)
|
|
92
|
+
finally:
|
|
93
|
+
elapsed_ms = round((time.monotonic() - started) * 1000)
|
|
94
|
+
print(
|
|
95
|
+
"callback_http_request "
|
|
96
|
+
f"status={status} elapsed_ms={elapsed_ms} error_type={error_type}",
|
|
97
|
+
file=sys.stderr,
|
|
98
|
+
flush=True,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
def log_message(self, format: str, *args: Any) -> None:
|
|
102
|
+
return
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def main() -> None:
|
|
106
|
+
port = int(os.environ.get("PORT", "8080"))
|
|
107
|
+
ThreadingHTTPServer(("0.0.0.0", port), Handler).serve_forever()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
if __name__ == "__main__":
|
|
111
|
+
main()
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
"""The sub-goals this agent can be checked on, shared by every scenario that needs one.
|
|
2
|
+
|
|
3
|
+
Defined once for the agent rather than restated per scenario, which is what makes results roll
|
|
4
|
+
up: the same sub-goal failing in seven of twelve scenarios is one sentence rather than seven.
|
|
5
|
+
|
|
6
|
+
``check`` is Python written by the harness. It is given what the run left behind and returns
|
|
7
|
+
nothing if the sub-goal held, or a sentence saying what was wrong. Code rather than a mini
|
|
8
|
+
language because an environment can be a database, a filesystem or a page, and a language
|
|
9
|
+
invented here would fit only the first.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import ast
|
|
15
|
+
import json
|
|
16
|
+
import textwrap
|
|
17
|
+
from collections.abc import Sequence
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from pydantic import BaseModel, Field
|
|
21
|
+
|
|
22
|
+
CATALOGUE = "sub_goals.json"
|
|
23
|
+
|
|
24
|
+
class SubGoal(BaseModel):
|
|
25
|
+
"""One named thing the agent can be checked on, shared across every scenario that needs it.
|
|
26
|
+
|
|
27
|
+
``check`` is Python, written by the harness. It is given what the run left behind and returns
|
|
28
|
+
nothing if the sub-goal held, or a sentence saying what was wrong. Code rather than a mini
|
|
29
|
+
language because an environment can be a database, a filesystem or a page, and a language
|
|
30
|
+
invented here would fit only the first.
|
|
31
|
+
|
|
32
|
+
``judged`` marks the ones nothing observable can settle — whether a refusal was explained,
|
|
33
|
+
whether a price was invented. Those go to a model, and are the exception.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
name: str
|
|
37
|
+
what: str = ""
|
|
38
|
+
check: str = ""
|
|
39
|
+
judged: str = ""
|
|
40
|
+
|
|
41
|
+
def deterministic(self) -> bool:
|
|
42
|
+
return bool(self.check.strip())
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class SuiteEval(BaseModel):
|
|
46
|
+
"""One built-in Future AGI eval applied to every compatible scenario."""
|
|
47
|
+
|
|
48
|
+
name: str
|
|
49
|
+
required_inputs: list[str] = Field(default_factory=lambda: ["conversation"])
|
|
50
|
+
minimum_score: float | None = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def default_suite_evals() -> list[SuiteEval]:
|
|
54
|
+
"""The two verified built-in evals initially run for every voice scenario."""
|
|
55
|
+
return [
|
|
56
|
+
SuiteEval(
|
|
57
|
+
name="customer_agent_task_completion",
|
|
58
|
+
required_inputs=["agent_prompt", "conversation"],
|
|
59
|
+
),
|
|
60
|
+
SuiteEval(
|
|
61
|
+
name="customer_agent_conversation_quality",
|
|
62
|
+
minimum_score=4,
|
|
63
|
+
),
|
|
64
|
+
]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class Catalogue(BaseModel):
|
|
68
|
+
"""Every sub-goal this agent has, defined once."""
|
|
69
|
+
|
|
70
|
+
sub_goals: list[SubGoal] = Field(default_factory=list)
|
|
71
|
+
# Deliberately separate from sub-goals: these assess every scenario, while a sub-goal only
|
|
72
|
+
# applies where a scenario names it.
|
|
73
|
+
suite_evals: list[SuiteEval] = Field(default_factory=default_suite_evals)
|
|
74
|
+
|
|
75
|
+
def named(self, name: str) -> SubGoal | None:
|
|
76
|
+
return next((one for one in self.sub_goals if one.name == name), None)
|
|
77
|
+
|
|
78
|
+
def names(self) -> set[str]:
|
|
79
|
+
return {one.name for one in self.sub_goals}
|
|
80
|
+
|
|
81
|
+
def suite_eval(self, name: str) -> SuiteEval | None:
|
|
82
|
+
return next((one for one in self.suite_evals if one.name == name), None)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def validate_suite_eval(suite_eval: SuiteEval) -> list[str]:
|
|
86
|
+
if not suite_eval.name.strip():
|
|
87
|
+
return ["no name"]
|
|
88
|
+
if not suite_eval.required_inputs:
|
|
89
|
+
return [f"{suite_eval.name}: no required inputs"]
|
|
90
|
+
return []
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def validate_sub_goal(sub_goal: SubGoal) -> list[str]:
|
|
94
|
+
"""Problems that make a sub-goal unusable.
|
|
95
|
+
|
|
96
|
+
A sub-goal that settles nothing is the expensive kind of wrong: every scenario referencing it
|
|
97
|
+
reports a result nobody should believe.
|
|
98
|
+
"""
|
|
99
|
+
problems: list[str] = []
|
|
100
|
+
if not sub_goal.name.strip():
|
|
101
|
+
problems.append("no name")
|
|
102
|
+
if not sub_goal.what.strip():
|
|
103
|
+
problems.append(f"{sub_goal.name}: no description of what it means")
|
|
104
|
+
if not sub_goal.check.strip() and not sub_goal.judged.strip():
|
|
105
|
+
problems.append(
|
|
106
|
+
f"{sub_goal.name}: settles nothing. Give a check in code, or say what a judge has "
|
|
107
|
+
"to decide and why nothing observable can settle it"
|
|
108
|
+
)
|
|
109
|
+
if sub_goal.check.strip() and "def check(" not in sub_goal.check:
|
|
110
|
+
problems.append(
|
|
111
|
+
f"{sub_goal.name}: a check must define check(world, calls) and return a problem as "
|
|
112
|
+
"a string, or None when the sub-goal held"
|
|
113
|
+
)
|
|
114
|
+
problems.extend(_presence_only_problems(sub_goal))
|
|
115
|
+
problems.extend(_judged_problems(sub_goal))
|
|
116
|
+
return problems
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# What a check has to touch to be about the outcome rather than about reaching a tool. `.arguments`
|
|
120
|
+
# is what the agent passed, `.result`/`.error` is what came back, and `world` is the state left
|
|
121
|
+
# behind. A check touching none of these can only be matching call names.
|
|
122
|
+
_OUTCOME_ATTRIBUTES = frozenset({"arguments", "args", "result", "error", "refused"})
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _reads_outcome(source: str) -> bool:
|
|
126
|
+
"""Whether a check reads what happened, rather than only that a call happened.
|
|
127
|
+
|
|
128
|
+
Parsed rather than string-matched. A substring test passes a check that merely mentions
|
|
129
|
+
``world`` in a comment or names an unused variable, and the whole point of this gate is that a
|
|
130
|
+
check which looks right and settles nothing is the expensive kind of wrong.
|
|
131
|
+
"""
|
|
132
|
+
try:
|
|
133
|
+
tree = ast.parse(textwrap.dedent(source))
|
|
134
|
+
except SyntaxError:
|
|
135
|
+
# An uncompilable check fails run_check with `broken` anyway, and reporting it as
|
|
136
|
+
# outcome-blind here would hide the real reason.
|
|
137
|
+
return True
|
|
138
|
+
|
|
139
|
+
for node in ast.walk(tree):
|
|
140
|
+
if isinstance(node, ast.Attribute) and node.attr in _OUTCOME_ATTRIBUTES:
|
|
141
|
+
return True
|
|
142
|
+
if isinstance(node, ast.Subscript):
|
|
143
|
+
# world.state()["orders"] parses as a Subscript over a Call; the attribute walk above
|
|
144
|
+
# catches `.state`, but a check handed a plain mapping is still reading state.
|
|
145
|
+
return True
|
|
146
|
+
if isinstance(node, ast.Name) and node.id == "world":
|
|
147
|
+
# Reading the world at all is reading state. The parameter itself is a Name in the
|
|
148
|
+
# signature, so only uses inside the body reach here.
|
|
149
|
+
for function in ast.walk(tree):
|
|
150
|
+
if isinstance(function, ast.FunctionDef) and function.name == "check":
|
|
151
|
+
names = {
|
|
152
|
+
inner.id
|
|
153
|
+
for statement in function.body
|
|
154
|
+
for inner in ast.walk(statement)
|
|
155
|
+
if isinstance(inner, ast.Name)
|
|
156
|
+
}
|
|
157
|
+
if "world" in names:
|
|
158
|
+
return True
|
|
159
|
+
return False
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _presence_only_problems(sub_goal: SubGoal) -> list[str]:
|
|
163
|
+
"""Refuse a check that only asks whether a tool was reached, rather than what it did."""
|
|
164
|
+
body = sub_goal.check
|
|
165
|
+
if not body.strip():
|
|
166
|
+
return []
|
|
167
|
+
if _reads_outcome(body):
|
|
168
|
+
return []
|
|
169
|
+
return [
|
|
170
|
+
f"{sub_goal.name}: the check only asks whether a tool was called, which any agent reaching "
|
|
171
|
+
"it passes and any agent doing the right thing another way fails. Assert the arguments it "
|
|
172
|
+
"was given, or the state the world was left in. Whether a call was ended is never a "
|
|
173
|
+
"sub-goal; what the agent did before stopping is"
|
|
174
|
+
]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def compares_to_a_value(source: str) -> bool:
|
|
178
|
+
"""Whether a check tests an argument against something specific, or only that it is non-empty.
|
|
179
|
+
|
|
180
|
+
This is the difference between "a reason was given" and "the reason was the right one". An
|
|
181
|
+
agent that mishears a name and proceeds confidently against the wrong record passes every
|
|
182
|
+
truthiness test: the argument is present, is a string, and is non-empty. Measured on a real
|
|
183
|
+
authored catalogue, five of six coded checks tested only truthiness.
|
|
184
|
+
|
|
185
|
+
Advisory rather than a refusal: hardening this would have refused five of those six, and an
|
|
186
|
+
authoring loop that cannot satisfy a gate fails the run instead of improving the check.
|
|
187
|
+
"""
|
|
188
|
+
try:
|
|
189
|
+
tree = ast.parse(textwrap.dedent(source))
|
|
190
|
+
except SyntaxError:
|
|
191
|
+
return False
|
|
192
|
+
for node in ast.walk(tree):
|
|
193
|
+
if not isinstance(node, ast.Compare):
|
|
194
|
+
continue
|
|
195
|
+
for op in node.ops:
|
|
196
|
+
if not isinstance(op, (ast.Eq, ast.NotEq, ast.In, ast.NotIn)):
|
|
197
|
+
continue
|
|
198
|
+
right = node.comparators[0] if node.comparators else None
|
|
199
|
+
# Comparing against None/""/0 is truthiness wearing a comparison's clothes, and
|
|
200
|
+
# matching on `c.name` is routing to the right call rather than judging its outcome.
|
|
201
|
+
if isinstance(right, ast.Constant) and right.value in (None, "", 0):
|
|
202
|
+
continue
|
|
203
|
+
if isinstance(node.left, ast.Attribute) and node.left.attr == "name":
|
|
204
|
+
continue
|
|
205
|
+
return True
|
|
206
|
+
return False
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def weak_check_advisory(sub_goal: SubGoal) -> str:
|
|
210
|
+
"""What to say about a check that reads the arguments but only tests that they are there."""
|
|
211
|
+
if not sub_goal.check.strip() or compares_to_a_value(sub_goal.check):
|
|
212
|
+
return ""
|
|
213
|
+
return (
|
|
214
|
+
f"{sub_goal.name}: the check reads the arguments but only tests that they are present and "
|
|
215
|
+
"non-empty. An agent that mishears a detail and acts confidently on the wrong one passes "
|
|
216
|
+
"that. Compare the value against what this scenario expected, or against the world row it "
|
|
217
|
+
"should match"
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _judged_problems(sub_goal: SubGoal) -> list[str]:
|
|
222
|
+
"""Hold a judged sub-goal to the reason it is judged.
|
|
223
|
+
|
|
224
|
+
The catalogue guidance already says a judge is the fallback, and nothing enforced it, so the
|
|
225
|
+
fallback became the default: one run reported six sub-goals judged rather than settled by code.
|
|
226
|
+
A judged sub-goal has to say what a model must decide and why nothing observable settles it,
|
|
227
|
+
because that sentence is the thing a reviewer can disagree with.
|
|
228
|
+
"""
|
|
229
|
+
judged = sub_goal.judged.strip()
|
|
230
|
+
if not judged or sub_goal.check.strip():
|
|
231
|
+
return []
|
|
232
|
+
# A word count is a crude proxy for "more than a restatement of the name". Four is where the
|
|
233
|
+
# real cases fall either side: "was it polite" is the name asked as a question and settles
|
|
234
|
+
# nothing, while "nothing observable shows tone" is terse and genuinely says why code cannot.
|
|
235
|
+
# An earlier threshold of eight rejected the second, which is a legitimate claim.
|
|
236
|
+
if len(judged.split()) < 4:
|
|
237
|
+
return [
|
|
238
|
+
f"{sub_goal.name}: judged, but does not say what a model has to decide and why nothing "
|
|
239
|
+
"observable settles it. Name the judgement and the reason code cannot make it, or "
|
|
240
|
+
"write a check"
|
|
241
|
+
]
|
|
242
|
+
return []
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def catalogue_problems(
|
|
246
|
+
sub_goals: Sequence[SubGoal], *, world_is_observable: bool = True
|
|
247
|
+
) -> list[str]:
|
|
248
|
+
"""Problems with the catalogue taken as a whole, rather than with one sub-goal.
|
|
249
|
+
|
|
250
|
+
Per-sub-goal validation cannot see the shape of the set, and the shape is what decides whether
|
|
251
|
+
a suite grades anything: a catalogue that is mostly judged reports opinions.
|
|
252
|
+
|
|
253
|
+
``world_is_observable`` is False for a target we cannot see into -- a conversational agent with
|
|
254
|
+
no executable tools and no state leaves nothing behind for a check to read, so judging is the
|
|
255
|
+
only thing available and is correct rather than lazy. It is never a way around writing a check
|
|
256
|
+
for a world that does have state.
|
|
257
|
+
"""
|
|
258
|
+
usable = [one for one in sub_goals if one.name.strip()]
|
|
259
|
+
if not usable or not world_is_observable:
|
|
260
|
+
return []
|
|
261
|
+
judged = [one for one in usable if not one.deterministic()]
|
|
262
|
+
if len(judged) * 2 > len(usable):
|
|
263
|
+
return [
|
|
264
|
+
f"{len(judged)} of {len(usable)} sub-goals are judged rather than settled by code. A "
|
|
265
|
+
"judge is the fallback, not the method: most of these are answerable from the "
|
|
266
|
+
"arguments the agent passed or the state it left. Rewrite the ones that are, and keep "
|
|
267
|
+
"judging only what nothing observable can settle"
|
|
268
|
+
]
|
|
269
|
+
return []
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def save_catalogue(catalogue: Catalogue, destination: Path) -> Path:
|
|
273
|
+
destination = Path(destination)
|
|
274
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
275
|
+
path = destination / CATALOGUE
|
|
276
|
+
path.write_text(
|
|
277
|
+
json.dumps(catalogue.model_dump(), indent=2, ensure_ascii=False),
|
|
278
|
+
encoding="utf-8",
|
|
279
|
+
)
|
|
280
|
+
return path
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def load_catalogue(destination: Path) -> Catalogue:
|
|
284
|
+
path = Path(destination) / CATALOGUE
|
|
285
|
+
if not path.exists():
|
|
286
|
+
return Catalogue()
|
|
287
|
+
return Catalogue.model_validate(json.loads(path.read_text(encoding="utf-8")))
|