agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1048 @@
|
|
|
1
|
+
"""A scenario: a delta on the base environment, and what must hold afterwards.
|
|
2
|
+
|
|
3
|
+
The base is built once — the world, the simulator's prompt, the catalogue of sub-goals. A
|
|
4
|
+
scenario changes a few values in that world, fills the prompt's slots, and names which sub-goals
|
|
5
|
+
must hold. It is not a template with values slotted into it; the harness writes each one.
|
|
6
|
+
|
|
7
|
+
It also carries a **solution**: what a correct agent would do. That is not decoration. It is what
|
|
8
|
+
proves, before the scenario is ever used, that the scenario can be passed at all and that its
|
|
9
|
+
checks are not vacuous — the two gates in ``prove.py``. Terminal-bench keeps its tasks honest the
|
|
10
|
+
same way, and it needs no model to do it.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import ast
|
|
16
|
+
import hashlib
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import re
|
|
20
|
+
from collections import Counter
|
|
21
|
+
from math import ceil
|
|
22
|
+
from typing import Any, ClassVar
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel, Field, model_validator
|
|
25
|
+
|
|
26
|
+
from .catalogue import Catalogue
|
|
27
|
+
from .simulator import variables_in
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# What a fixture's `origin` may say, and which of those claim the scenario creates data itself.
|
|
31
|
+
FIXTURE_ORIGINS = ("seed", "generated", "mixed")
|
|
32
|
+
ORIGINS_THAT_CREATE = ("generated", "mixed")
|
|
33
|
+
|
|
34
|
+
# For an outbound call, how much the person already knows about why they are being rung. The order
|
|
35
|
+
# is the axis: told to expect it, half remembers, no idea at all. Named once so the schema a writer
|
|
36
|
+
# is offered, the suite's spread rule and the caller's own briefing cannot drift apart.
|
|
37
|
+
CALLER_AWARENESS = ("expecting", "partial", "unaware")
|
|
38
|
+
LEAST_AWARE = "unaware"
|
|
39
|
+
|
|
40
|
+
# Who picked up an outbound call. Empty means a person; "voicemail" means nobody is on the line.
|
|
41
|
+
ANSWERED_BY = ("person", "voicemail")
|
|
42
|
+
VOICEMAIL = "voicemail"
|
|
43
|
+
|
|
44
|
+
# Which kind of mailbox answered: what the greeting says, and whether a tone follows it.
|
|
45
|
+
VOICEMAIL_STYLES = ("personal", "carrier", "operator", "full")
|
|
46
|
+
DEFAULT_VOICEMAIL_STYLE = "personal"
|
|
47
|
+
|
|
48
|
+
# The share of a suite a mailbox may occupy: a ceiling with no floor, since none is legitimate.
|
|
49
|
+
RARE_CONDITION_SHARE = 0.05
|
|
50
|
+
|
|
51
|
+
# The switch that removes mailboxes from a run altogether, for when they are not wanted at all
|
|
52
|
+
# rather than merely kept rare.
|
|
53
|
+
VOICEMAIL_SWITCH = "ALK_VOICEMAIL_SCENARIOS"
|
|
54
|
+
_OFF = ("0", "off", "false", "no")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def voicemail_enabled() -> bool:
|
|
58
|
+
"""Whether this run may write or place a call a mailbox answers. On unless the switch says no."""
|
|
59
|
+
return os.environ.get(VOICEMAIL_SWITCH, "1").strip().lower() not in _OFF
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Step(BaseModel):
|
|
63
|
+
"""One action in a reference solution."""
|
|
64
|
+
|
|
65
|
+
tool: str
|
|
66
|
+
arguments: dict[str, Any] = Field(default_factory=dict)
|
|
67
|
+
# Source-backed agents often add trusted session state between the model-facing function
|
|
68
|
+
# and the dependency API: internal identifiers, resolved lookups, priced results, and similar
|
|
69
|
+
# must never be exposed as arguments the model supposedly chose. A reference proof still
|
|
70
|
+
# has to drive the real dependency so its database effects can be checked, so it may carry
|
|
71
|
+
# that dependency payload separately. Agent runs never read this field.
|
|
72
|
+
environment_arguments: dict[str, Any] = Field(default_factory=dict)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class Persona(BaseModel):
|
|
76
|
+
"""The simulated caller, in the same shape used by existing voice scenarios.
|
|
77
|
+
|
|
78
|
+
A persona controls how the caller pursues a scenario's task. The task itself remains on
|
|
79
|
+
``Scenario.instruction`` so the harness can vary either one without conflating them.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
name: str = ""
|
|
83
|
+
gender: str = ""
|
|
84
|
+
age_group: str = ""
|
|
85
|
+
occupation: str = ""
|
|
86
|
+
location: str = ""
|
|
87
|
+
personality: str = ""
|
|
88
|
+
communication_style: str = ""
|
|
89
|
+
# The first thing this person actually says. Voice agents often greet immediately; leaving
|
|
90
|
+
# this to the simulator model produced generic "Hello?" turns and avoidable silence races.
|
|
91
|
+
initial_message: str = ""
|
|
92
|
+
keywords: list[str] = Field(default_factory=list)
|
|
93
|
+
languages: list[str] = Field(default_factory=list)
|
|
94
|
+
accent: str = ""
|
|
95
|
+
multilingual: bool = False
|
|
96
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
97
|
+
# Optional deterministic voice policy for transactional scenarios. It keeps
|
|
98
|
+
# caller facts realistic and varied while avoiding LLM role drift during a
|
|
99
|
+
# long tool-heavy phone flow.
|
|
100
|
+
scripted_caller: dict[str, Any] | None = None
|
|
101
|
+
|
|
102
|
+
def described(self) -> bool:
|
|
103
|
+
return bool(
|
|
104
|
+
self.name
|
|
105
|
+
or self.gender
|
|
106
|
+
or self.age_group
|
|
107
|
+
or self.occupation
|
|
108
|
+
or self.location
|
|
109
|
+
or self.personality
|
|
110
|
+
or self.communication_style
|
|
111
|
+
or self.keywords
|
|
112
|
+
or self.languages
|
|
113
|
+
or self.accent
|
|
114
|
+
or self.metadata
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
def missing_profile_fields(self) -> list[str]:
|
|
118
|
+
"""The minimum needed for a scenario to exercise caller variation intentionally."""
|
|
119
|
+
missing = [
|
|
120
|
+
name
|
|
121
|
+
for name, value in (
|
|
122
|
+
("name", self.name),
|
|
123
|
+
("personality", self.personality),
|
|
124
|
+
("communication_style", self.communication_style),
|
|
125
|
+
("initial_message", self.initial_message),
|
|
126
|
+
("accent", self.accent),
|
|
127
|
+
)
|
|
128
|
+
if not value.strip()
|
|
129
|
+
]
|
|
130
|
+
if not self.languages:
|
|
131
|
+
missing.append("languages")
|
|
132
|
+
if not self.keywords:
|
|
133
|
+
missing.append("keywords")
|
|
134
|
+
return missing
|
|
135
|
+
|
|
136
|
+
def format_persona(self) -> str:
|
|
137
|
+
"""A stable, human-readable profile the simulator can consistently embody."""
|
|
138
|
+
parts = []
|
|
139
|
+
identity = []
|
|
140
|
+
for label, value in (
|
|
141
|
+
("Name", self.name),
|
|
142
|
+
("Gender", self.gender),
|
|
143
|
+
("Age Group", self.age_group),
|
|
144
|
+
("Occupation", self.occupation),
|
|
145
|
+
("Location", self.location),
|
|
146
|
+
):
|
|
147
|
+
if value:
|
|
148
|
+
identity.append(f"- {label}: {value}")
|
|
149
|
+
if identity:
|
|
150
|
+
parts.append("# YOUR IDENTITY\n\n" + "\n".join(identity))
|
|
151
|
+
|
|
152
|
+
behavior = []
|
|
153
|
+
if self.personality:
|
|
154
|
+
behavior.append(f"- Personality: {self.personality}")
|
|
155
|
+
if self.communication_style:
|
|
156
|
+
behavior.append(f"- Communication Style: {self.communication_style}")
|
|
157
|
+
if self.keywords:
|
|
158
|
+
behavior.append("- Key Traits: " + ", ".join(self.keywords))
|
|
159
|
+
if behavior:
|
|
160
|
+
parts.append("# YOUR PERSONALITY & COMMUNICATION\n\n" + "\n".join(behavior))
|
|
161
|
+
|
|
162
|
+
speech = []
|
|
163
|
+
if self.languages:
|
|
164
|
+
speech.append("- Language(s): " + ", ".join(self.languages))
|
|
165
|
+
if self.accent:
|
|
166
|
+
speech.append(f"- Accent: {self.accent}")
|
|
167
|
+
if self.multilingual:
|
|
168
|
+
speech.append(
|
|
169
|
+
"- Switch languages naturally when the conversation calls for it."
|
|
170
|
+
)
|
|
171
|
+
if speech:
|
|
172
|
+
parts.append("# LANGUAGE & SPEECH PATTERNS\n\n" + "\n".join(speech))
|
|
173
|
+
|
|
174
|
+
if self.metadata:
|
|
175
|
+
characteristics = [
|
|
176
|
+
f"- {key.replace('_', ' ').title()}: {value}"
|
|
177
|
+
for key, value in self.metadata.items()
|
|
178
|
+
]
|
|
179
|
+
parts.append(
|
|
180
|
+
"# ADDITIONAL CHARACTERISTICS\n\n" + "\n".join(characteristics)
|
|
181
|
+
)
|
|
182
|
+
return "\n".join(parts)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _slug(name: str) -> str:
|
|
186
|
+
"""An ASCII key for ``name``, safe to send as a header value.
|
|
187
|
+
|
|
188
|
+
Falls back to a digest rather than an empty string: an empty key would collapse every
|
|
189
|
+
scenario in a job onto one idempotency key on the receiving side.
|
|
190
|
+
"""
|
|
191
|
+
cleaned = re.sub(r"[^a-z0-9]+", "-", (name or "").strip().lower()).strip("-")
|
|
192
|
+
return cleaned or "scenario-" + hashlib.sha256(name.encode()).hexdigest()[:12]
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _decided_by(name: str) -> bool:
|
|
196
|
+
"""Whether this scenario is noisy, decided by its name so a rerun decides the same."""
|
|
197
|
+
return hashlib.sha256((name or "").encode()).digest()[0] % 2 == 0
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
class Scenario(BaseModel):
|
|
201
|
+
"""One test: what changes, what is asked, what a correct agent does, what must hold."""
|
|
202
|
+
|
|
203
|
+
name: str
|
|
204
|
+
# How this scenario is identified on the wire. Derived from ``name``, which is already unique
|
|
205
|
+
# across a suite and already a slug because it is the folder name. It ships as a header, so
|
|
206
|
+
# anything outside ASCII is dropped and an empty result falls back to a digest.
|
|
207
|
+
scenario_key: str = ""
|
|
208
|
+
# Assigned by the platform when the scenario is pre-allocated. Never written here.
|
|
209
|
+
scenario_id: str = ""
|
|
210
|
+
use_case: str = ""
|
|
211
|
+
# What makes this row different from its siblings in the same use case. Coverage is counted
|
|
212
|
+
# on the pair, so a use case can carry many scenarios without any reading as a duplicate.
|
|
213
|
+
branch: str = ""
|
|
214
|
+
tests: str = ""
|
|
215
|
+
|
|
216
|
+
# What this scenario changes about the world after it is reset, as code: a file defining
|
|
217
|
+
# ``setup(world)``. Rows in a table were enough while every world was a database, and they
|
|
218
|
+
# are not enough now — a scenario may need a service to start returning errors, a file to be
|
|
219
|
+
# missing, a queue to be backed up. Code can express all of that; a table of rows cannot.
|
|
220
|
+
setup_code: str = ""
|
|
221
|
+
|
|
222
|
+
# Whether the world is actually ready for this scenario, as code: a file defining
|
|
223
|
+
# ``ready(world)`` that answers with nothing when the world holds what this scenario
|
|
224
|
+
# presumes, or a sentence saying what is missing.
|
|
225
|
+
#
|
|
226
|
+
# This is the precondition, and it is the difference between a real finding and a wasted
|
|
227
|
+
# run: a scenario about the last five chocolates is only a test of the agent if there really
|
|
228
|
+
# are five. Otherwise the agent fails for something we got wrong, and it looks like the
|
|
229
|
+
# agent's fault.
|
|
230
|
+
ready_code: str = ""
|
|
231
|
+
|
|
232
|
+
# The task. For a conversational agent it fills the simulator prompt's instruction slot; for
|
|
233
|
+
# a browser or coding agent it goes to the agent directly.
|
|
234
|
+
instruction: str = ""
|
|
235
|
+
# Who is making the request. This is deliberately separate from the task so a caller's
|
|
236
|
+
# communication needs do not get buried in an unstructured instruction.
|
|
237
|
+
persona: Persona | None = None
|
|
238
|
+
# Anything else that prompt asks for, by slot name.
|
|
239
|
+
variables: dict[str, str] = Field(default_factory=dict)
|
|
240
|
+
# A readable declaration of which data makes this scenario real. ``setup_code`` remains the
|
|
241
|
+
# executable delta; this is the index a person and the UI can inspect without reverse-
|
|
242
|
+
# engineering Python. Typical keys are origin (seed/generated/mixed), identity, credentials,
|
|
243
|
+
# location and account_state. It is intentionally open-ended across agent domains.
|
|
244
|
+
fixture: dict[str, Any] = Field(default_factory=dict)
|
|
245
|
+
|
|
246
|
+
# What a correct agent would do. Run by the gates, never by the agent under test.
|
|
247
|
+
solution: list[Step] = Field(default_factory=list)
|
|
248
|
+
|
|
249
|
+
# Which entries of the shared catalogue must hold. Named, not restated, so results roll up
|
|
250
|
+
# across the suite: the same sub-goal failing in seven of twelve scenarios is one sentence.
|
|
251
|
+
sub_goals: list[str] = Field(default_factory=list)
|
|
252
|
+
|
|
253
|
+
max_turns: int = 10
|
|
254
|
+
|
|
255
|
+
# Where this call is made from. A string names the place ("street", "vehicle", "retail"), and
|
|
256
|
+
# True asks for noise while leaving the place to the fixture. Left unset it is decided from
|
|
257
|
+
# the name, so a suite still covers both conditions but the same suite decides the same way
|
|
258
|
+
# twice; a coin flip here made a seeded run unreproducible.
|
|
259
|
+
background_noise: bool | str = ""
|
|
260
|
+
|
|
261
|
+
# Whether the agent placed this call or answered it. Voice only: a chat is always started by
|
|
262
|
+
# the person, so it stays inbound. An outbound caller has no opening request to make, which is
|
|
263
|
+
# a different test of the agent rather than the same one with a reworded greeting.
|
|
264
|
+
# Empty means defer to the run and then to the contract, which is where the agent's own
|
|
265
|
+
# direction was identified. Defaulting it to "inbound" here would be written into the saved
|
|
266
|
+
# document and silently outrank both of them.
|
|
267
|
+
call_direction: str = ""
|
|
268
|
+
# For an outbound call, how much this person already knows about why they are being rung:
|
|
269
|
+
# "expecting", "partial" or "unaware". Unset means unaware, the case the agent must work
|
|
270
|
+
# hardest for.
|
|
271
|
+
caller_awareness: str = ""
|
|
272
|
+
# Who answered. Outbound only: a mailbox cannot answer a call the person placed themselves.
|
|
273
|
+
answered_by: str = ""
|
|
274
|
+
# Which kind of mailbox answered. Only read where answered_by is "voicemail"; empty is personal.
|
|
275
|
+
voicemail_style: str = ""
|
|
276
|
+
|
|
277
|
+
# Slots the caller filled by the run rather than by the scenario. Listed so a template that
|
|
278
|
+
# uses one is not rejected as unfillable at write time.
|
|
279
|
+
RUNTIME_SLOTS: ClassVar[tuple[str, ...]] = ("channel", "situation")
|
|
280
|
+
|
|
281
|
+
@model_validator(mode="after")
|
|
282
|
+
def _identify(self) -> "Scenario":
|
|
283
|
+
if not self.scenario_key:
|
|
284
|
+
self.scenario_key = _slug(self.name)
|
|
285
|
+
if self.background_noise == "":
|
|
286
|
+
self.background_noise = _decided_by(self.name)
|
|
287
|
+
return self
|
|
288
|
+
|
|
289
|
+
def slots(self) -> dict[str, str]:
|
|
290
|
+
"""Every value this scenario offers the simulator prompt."""
|
|
291
|
+
persona = {"persona": self.persona.format_persona()} if self.persona else {}
|
|
292
|
+
runtime = {name: "" for name in self.RUNTIME_SLOTS}
|
|
293
|
+
return {
|
|
294
|
+
"instruction": self.instruction,
|
|
295
|
+
**runtime,
|
|
296
|
+
**self.variables,
|
|
297
|
+
**persona,
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def validate_scenario(
|
|
302
|
+
scenario: Scenario,
|
|
303
|
+
catalogue: Catalogue,
|
|
304
|
+
world_state: dict[str, list[dict[str, Any]]],
|
|
305
|
+
simulator_prompt: str = "",
|
|
306
|
+
*,
|
|
307
|
+
allow_empty_solution: bool = False,
|
|
308
|
+
) -> list[str]:
|
|
309
|
+
"""Problems that make a scenario unusable, found without running anything.
|
|
310
|
+
|
|
311
|
+
Whether it can actually be passed is a different question, and no amount of reading settles
|
|
312
|
+
it. That is what the gates are for.
|
|
313
|
+
"""
|
|
314
|
+
problems: list[str] = []
|
|
315
|
+
if not scenario.name.strip():
|
|
316
|
+
problems.append("no name")
|
|
317
|
+
if not scenario.instruction.strip():
|
|
318
|
+
problems.append("no instruction: there is nothing for the run to be about")
|
|
319
|
+
if scenario.persona is not None and not scenario.persona.described():
|
|
320
|
+
problems.append("persona has no details")
|
|
321
|
+
elif scenario.persona is not None and (
|
|
322
|
+
missing := scenario.persona.missing_profile_fields()
|
|
323
|
+
):
|
|
324
|
+
problems.append("persona is incomplete: " + ", ".join(missing))
|
|
325
|
+
elif scenario.persona is not None:
|
|
326
|
+
# A persona in words of its own renders fine and then does nothing: no behaviour guidance
|
|
327
|
+
# attaches, and the accent it names selects no voice.
|
|
328
|
+
from .persona_guides import unrecognised
|
|
329
|
+
|
|
330
|
+
problems.extend(unrecognised(scenario.persona.model_dump()))
|
|
331
|
+
if not scenario.sub_goals:
|
|
332
|
+
problems.append(
|
|
333
|
+
"no sub_goals: nothing would be graded. Name the entries of the catalogue this "
|
|
334
|
+
"scenario is meant to exercise"
|
|
335
|
+
)
|
|
336
|
+
if world_state and not scenario.fixture:
|
|
337
|
+
problems.append(
|
|
338
|
+
"no fixture manifest: declare the seed/generated/mixed data this scenario relies on"
|
|
339
|
+
)
|
|
340
|
+
elif scenario.fixture and str(
|
|
341
|
+
scenario.fixture.get("origin") or ""
|
|
342
|
+
).lower() not in set(FIXTURE_ORIGINS):
|
|
343
|
+
problems.append(
|
|
344
|
+
"fixture.origin must be "
|
|
345
|
+
+ ", ".join(FIXTURE_ORIGINS[:-1])
|
|
346
|
+
+ f", or {FIXTURE_ORIGINS[-1]}"
|
|
347
|
+
)
|
|
348
|
+
elif (
|
|
349
|
+
scenario.fixture
|
|
350
|
+
and str(scenario.fixture.get("origin") or "").lower()
|
|
351
|
+
in set(ORIGINS_THAT_CREATE)
|
|
352
|
+
and not (scenario.setup_code or "").strip()
|
|
353
|
+
):
|
|
354
|
+
# A fixture claiming data it never creates is the whole class of scenario that names a value
|
|
355
|
+
# the scenario reads as self-sufficient, the world has none of it, and the agent has nothing
|
|
356
|
+
# to answer with. Caught here because it is provable from the document alone.
|
|
357
|
+
problems.append(
|
|
358
|
+
f"fixture.origin is {scenario.fixture.get('origin')!r}, which claims this scenario "
|
|
359
|
+
"creates data, but setup_code is empty. Either seed everything the fixture names, or "
|
|
360
|
+
"declare origin 'seed' and use only records that already exist"
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
unknown = sorted(set(scenario.sub_goals) - catalogue.names())
|
|
364
|
+
if unknown:
|
|
365
|
+
problems.append(
|
|
366
|
+
f"sub_goals not in the catalogue: {', '.join(unknown)}. Use the shared names, or add "
|
|
367
|
+
f"them to the catalogue first. It has: {', '.join(sorted(catalogue.names())) or 'none'}"
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
# setup_code and ready_code are not read here. Whether they work is not a question reading
|
|
371
|
+
# them can answer, and running them is exactly what the first gate does.
|
|
372
|
+
if scenario.setup_code.strip() and "def setup(" not in scenario.setup_code:
|
|
373
|
+
problems.append("setup_code must define setup(world)")
|
|
374
|
+
if scenario.ready_code.strip() and "def ready(" not in scenario.ready_code:
|
|
375
|
+
problems.append("ready_code must define ready(world)")
|
|
376
|
+
|
|
377
|
+
if simulator_prompt:
|
|
378
|
+
unfilled = sorted(variables_in(simulator_prompt) - set(scenario.slots()))
|
|
379
|
+
if unfilled:
|
|
380
|
+
problems.append(
|
|
381
|
+
f"the simulator prompt asks for {', '.join(unfilled)}, which this scenario does "
|
|
382
|
+
"not supply. An unfilled slot reaches the caller verbatim"
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
if not scenario.solution and not allow_empty_solution:
|
|
386
|
+
problems.append(
|
|
387
|
+
"no solution: without the actions a correct agent would take, there is no way to "
|
|
388
|
+
"show this scenario can be passed at all"
|
|
389
|
+
)
|
|
390
|
+
problems.extend(fixture_problems(scenario))
|
|
391
|
+
problems.extend(answered_by_problems(scenario))
|
|
392
|
+
problems.extend(voicemail_style_problems(scenario))
|
|
393
|
+
problems.extend(voicemail_sub_goal_problems(scenario, catalogue))
|
|
394
|
+
problems.extend(_world_credential_problems(scenario, world_state))
|
|
395
|
+
problems.extend(self_sufficiency_problems(scenario))
|
|
396
|
+
problems.extend(alignment_problems(scenario, world_state))
|
|
397
|
+
problems.extend(hollow_scenario_problems(scenario))
|
|
398
|
+
problems.extend(naming_problems(scenario))
|
|
399
|
+
return problems
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def answered_by_problems(scenario: Scenario) -> list[str]:
|
|
403
|
+
"""Whether what answered could have: a mailbox only exists on a call the agent placed."""
|
|
404
|
+
chosen = str(scenario.answered_by or "").strip().lower()
|
|
405
|
+
if not chosen:
|
|
406
|
+
return []
|
|
407
|
+
if chosen not in set(ANSWERED_BY):
|
|
408
|
+
return [
|
|
409
|
+
"answered_by must be "
|
|
410
|
+
+ ", ".join(ANSWERED_BY)
|
|
411
|
+
+ f", not {scenario.answered_by!r}"
|
|
412
|
+
]
|
|
413
|
+
if chosen != VOICEMAIL:
|
|
414
|
+
return []
|
|
415
|
+
if not voicemail_enabled():
|
|
416
|
+
return [
|
|
417
|
+
f"answered_by {VOICEMAIL!r} is turned off for this run, so write a scenario somebody "
|
|
418
|
+
"answers instead"
|
|
419
|
+
]
|
|
420
|
+
if str(scenario.call_direction or "").strip().lower() != "outbound":
|
|
421
|
+
return [
|
|
422
|
+
"answered_by is 'voicemail', which only happens on a call the agent placed, so this "
|
|
423
|
+
"scenario must state call_direction 'outbound' itself. Left unset, the direction is "
|
|
424
|
+
"taken from the contract and a mailbox would be answering a call the person dialled"
|
|
425
|
+
]
|
|
426
|
+
return []
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def voicemail_sub_goal_problems(scenario: Scenario, catalogue: Catalogue) -> list[str]:
|
|
430
|
+
"""Whether this mailbox scenario asks for something a mailbox call can produce.
|
|
431
|
+
|
|
432
|
+
A sub-goal needing a tool call cannot hold when nobody answers, so it would fail a correctly
|
|
433
|
+
handled mailbox. Read from the check rather than the name, since the check is what decides.
|
|
434
|
+
"""
|
|
435
|
+
if str(scenario.answered_by or "").strip().lower() != VOICEMAIL:
|
|
436
|
+
return []
|
|
437
|
+
problems: list[str] = []
|
|
438
|
+
for name in scenario.sub_goals:
|
|
439
|
+
sub_goal = catalogue.named(name)
|
|
440
|
+
if sub_goal is None:
|
|
441
|
+
continue
|
|
442
|
+
check = " ".join(str(sub_goal.check or "").split())
|
|
443
|
+
if not check or "calls" not in check:
|
|
444
|
+
continue
|
|
445
|
+
# Any negative test over a filtered call list, which is the shape writers produce.
|
|
446
|
+
needs_a_call = (
|
|
447
|
+
"not any(" in check
|
|
448
|
+
or "not called" in check
|
|
449
|
+
or "calls == []" in check
|
|
450
|
+
or re.search(r"if not [A-Za-z_][A-Za-z0-9_]*\s*:", check) is not None
|
|
451
|
+
or re.search(r"len\([A-Za-z_][A-Za-z0-9_]*\) *== *0", check) is not None
|
|
452
|
+
)
|
|
453
|
+
if needs_a_call:
|
|
454
|
+
problems.append(
|
|
455
|
+
f"sub_goal {name!r} fails when a tool was not called, and on this scenario a mailbox "
|
|
456
|
+
"answers, so the agent never gets the turn that leads it to call anything. Ask for "
|
|
457
|
+
"what the agent can do with nobody on the line: that it recognised a machine, that "
|
|
458
|
+
"the message it left says who is calling and why, that it stopped instead of asking "
|
|
459
|
+
"questions. A sub-goal needing an answer marks a correctly handled mailbox as failed"
|
|
460
|
+
)
|
|
461
|
+
return problems
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def voicemail_style_problems(scenario: Scenario) -> list[str]:
|
|
465
|
+
"""Whether the named style exists, and whether a mailbox answered to play it at all."""
|
|
466
|
+
chosen = str(scenario.voicemail_style or "").strip().lower()
|
|
467
|
+
if not chosen:
|
|
468
|
+
return []
|
|
469
|
+
if chosen not in set(VOICEMAIL_STYLES):
|
|
470
|
+
return [
|
|
471
|
+
"voicemail_style must be "
|
|
472
|
+
+ ", ".join(VOICEMAIL_STYLES)
|
|
473
|
+
+ f", not {scenario.voicemail_style!r}"
|
|
474
|
+
]
|
|
475
|
+
if str(scenario.answered_by or "").strip().lower() != VOICEMAIL:
|
|
476
|
+
return [
|
|
477
|
+
f"voicemail_style is {chosen!r} but answered_by is not 'voicemail', so no mailbox "
|
|
478
|
+
"answers and nothing plays it. State answered_by 'voicemail', or leave the style out"
|
|
479
|
+
]
|
|
480
|
+
return []
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def _world_credential_problems(
|
|
484
|
+
scenario: Scenario, world_state: dict[str, list[dict[str, Any]]]
|
|
485
|
+
) -> list[str]:
|
|
486
|
+
"""Reject caller credentials paired with the wrong world identity.
|
|
487
|
+
|
|
488
|
+
Hosted source authoring cannot execute a repository's runtime-owned tools until the sealed
|
|
489
|
+
bundle is provisioned. Static scenario validation must therefore catch identity-bound test
|
|
490
|
+
credentials that a deferred reference rehearsal cannot. The matching is deliberately
|
|
491
|
+
schema-shaped rather than application-shaped: any collection containing a phone-like
|
|
492
|
+
identity and an OTP/verification code participates.
|
|
493
|
+
"""
|
|
494
|
+
credentials: dict[str, set[str]] = {}
|
|
495
|
+
for collection, rows in world_state.items():
|
|
496
|
+
if not any(token in collection.lower() for token in ("otp", "verification")):
|
|
497
|
+
continue
|
|
498
|
+
for row in rows:
|
|
499
|
+
phone = next(
|
|
500
|
+
(
|
|
501
|
+
str(value)
|
|
502
|
+
for key, value in row.items()
|
|
503
|
+
if "phone" in str(key).lower() and value not in (None, "")
|
|
504
|
+
),
|
|
505
|
+
"",
|
|
506
|
+
)
|
|
507
|
+
code = next(
|
|
508
|
+
(
|
|
509
|
+
str(value).replace(" ", "")
|
|
510
|
+
for key, value in row.items()
|
|
511
|
+
if ("code" in str(key).lower() or "otp" in str(key).lower())
|
|
512
|
+
and re.fullmatch(r"\d{4,10}", str(value).replace(" ", ""))
|
|
513
|
+
),
|
|
514
|
+
"",
|
|
515
|
+
)
|
|
516
|
+
if phone and code:
|
|
517
|
+
credentials.setdefault(phone, set()).add(code)
|
|
518
|
+
|
|
519
|
+
fixture = scenario.fixture or {}
|
|
520
|
+
phones: set[str] = set()
|
|
521
|
+
codes: set[str] = set()
|
|
522
|
+
|
|
523
|
+
def collect(value: Any, key: str = "") -> None:
|
|
524
|
+
if isinstance(value, dict):
|
|
525
|
+
for child, item in value.items():
|
|
526
|
+
collect(item, str(child))
|
|
527
|
+
elif isinstance(value, list):
|
|
528
|
+
for item in value:
|
|
529
|
+
collect(item, key)
|
|
530
|
+
elif "phone" in key.lower() and value not in (None, ""):
|
|
531
|
+
phones.add(str(value))
|
|
532
|
+
elif "otp" in key.lower() or key.lower() in {"code", "verification_code"}:
|
|
533
|
+
candidate = str(value).replace(" ", "")
|
|
534
|
+
if re.fullmatch(r"\d{4,10}", candidate):
|
|
535
|
+
codes.add(candidate)
|
|
536
|
+
|
|
537
|
+
collect(fixture)
|
|
538
|
+
if scenario.persona is not None:
|
|
539
|
+
collect(scenario.persona.metadata)
|
|
540
|
+
collect(scenario.persona.scripted_caller or {})
|
|
541
|
+
for step in scenario.solution:
|
|
542
|
+
if "verify" in step.tool.lower() or "otp" in step.tool.lower():
|
|
543
|
+
collect(step.arguments)
|
|
544
|
+
|
|
545
|
+
problems: list[str] = []
|
|
546
|
+
for phone in sorted(phones & credentials.keys()):
|
|
547
|
+
wrong = codes - credentials[phone]
|
|
548
|
+
if codes and wrong and not (codes & credentials[phone]):
|
|
549
|
+
problems.append(
|
|
550
|
+
"verification credential does not belong to the scenario caller "
|
|
551
|
+
f"{phone}; inspect the world and use that identity's code"
|
|
552
|
+
)
|
|
553
|
+
return problems
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def contract_sequence_problems(
|
|
557
|
+
scenario: Scenario, hard_constraints: list[str]
|
|
558
|
+
) -> list[str]:
|
|
559
|
+
"""Catch reference solutions that hide required same-call state in a fixture.
|
|
560
|
+
|
|
561
|
+
A dependency can accept a pre-seeded identifier even when the public agent API cannot. For
|
|
562
|
+
a rule such as ``cancel_ride requires a booking_ref from this call``, require a producer
|
|
563
|
+
(``book_ride``) earlier in the same reference solution instead of allowing setup code or
|
|
564
|
+
environment-only arguments to make an impossible scenario look solvable.
|
|
565
|
+
"""
|
|
566
|
+
problems: list[str] = []
|
|
567
|
+
names = [step.tool for step in scenario.solution]
|
|
568
|
+
pattern = re.compile(
|
|
569
|
+
r"\b(?P<consumer>[a-z][a-z0-9_]*)\b\s+requires\b.*?\b"
|
|
570
|
+
r"(?P<resource>[a-z][a-z0-9_]*(?:_id|_ref))\s+from this call\b",
|
|
571
|
+
re.IGNORECASE,
|
|
572
|
+
)
|
|
573
|
+
for constraint in hard_constraints:
|
|
574
|
+
found = pattern.search(constraint)
|
|
575
|
+
if found is None:
|
|
576
|
+
continue
|
|
577
|
+
consumer = found.group("consumer").lower()
|
|
578
|
+
lowered = [name.lower() for name in names]
|
|
579
|
+
if consumer not in lowered:
|
|
580
|
+
continue
|
|
581
|
+
resource = re.sub(r"_(?:id|ref)$", "", found.group("resource").lower())
|
|
582
|
+
stems = {resource, resource.removesuffix("ing")}
|
|
583
|
+
before = lowered[: lowered.index(consumer)]
|
|
584
|
+
produced = any(
|
|
585
|
+
any(stem and stem in tool for stem in stems)
|
|
586
|
+
and not tool.startswith(("get_", "list_", "find_", "cancel_"))
|
|
587
|
+
for tool in before
|
|
588
|
+
)
|
|
589
|
+
if not produced:
|
|
590
|
+
problems.append(
|
|
591
|
+
f"{consumer} requires {found.group('resource')} from this call, but the "
|
|
592
|
+
"reference solution does not create it first; do not hide it in setup or "
|
|
593
|
+
"environment_arguments"
|
|
594
|
+
)
|
|
595
|
+
return problems
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
_WEAK_CODES = {
|
|
599
|
+
"000000",
|
|
600
|
+
"111111",
|
|
601
|
+
"222222",
|
|
602
|
+
"333333",
|
|
603
|
+
"444444",
|
|
604
|
+
"555555",
|
|
605
|
+
"666666",
|
|
606
|
+
"777777",
|
|
607
|
+
"888888",
|
|
608
|
+
"999999",
|
|
609
|
+
"012345",
|
|
610
|
+
"123456",
|
|
611
|
+
"234567",
|
|
612
|
+
"345678",
|
|
613
|
+
"456789",
|
|
614
|
+
"987654",
|
|
615
|
+
"876543",
|
|
616
|
+
"765432",
|
|
617
|
+
"654321",
|
|
618
|
+
}
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
def _six_digit_values(scenario: Scenario) -> list[str]:
|
|
622
|
+
"""Likely one-time codes declared by a scenario, without treating phone digits as OTPs."""
|
|
623
|
+
found: list[str] = []
|
|
624
|
+
|
|
625
|
+
def walk(value: Any, key: str = "") -> None:
|
|
626
|
+
if isinstance(value, dict):
|
|
627
|
+
for child, item in value.items():
|
|
628
|
+
walk(item, str(child))
|
|
629
|
+
elif isinstance(value, list):
|
|
630
|
+
for item in value:
|
|
631
|
+
walk(item, key)
|
|
632
|
+
elif "otp" in key.lower() or key.lower() in {"code", "verification_code"}:
|
|
633
|
+
found.extend(re.findall(r"(?<!\d)\d{6}(?!\d)", str(value)))
|
|
634
|
+
|
|
635
|
+
walk(scenario.fixture)
|
|
636
|
+
if scenario.persona:
|
|
637
|
+
walk(scenario.persona.metadata)
|
|
638
|
+
walk(scenario.persona.scripted_caller or {})
|
|
639
|
+
for step in scenario.solution:
|
|
640
|
+
walk(step.arguments)
|
|
641
|
+
walk(step.environment_arguments)
|
|
642
|
+
# Setup is code, so key-aware traversal is unavailable. Restrict matches to a nearby field
|
|
643
|
+
# name instead of collecting six digits from a phone number or an unrelated identifier.
|
|
644
|
+
found.extend(
|
|
645
|
+
match.group(1)
|
|
646
|
+
for match in re.finditer(
|
|
647
|
+
r"(?:otp|verification[_ ]?code|['\"]code['\"])[^\n]{0,80}?(?<!\d)(\d{6})(?!\d)",
|
|
648
|
+
scenario.setup_code,
|
|
649
|
+
flags=re.IGNORECASE,
|
|
650
|
+
)
|
|
651
|
+
)
|
|
652
|
+
return found
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
# A value the instruction hands the caller so they can say it back: a code, a reference, an account
|
|
656
|
+
# number, an id. Deliberately not named after any one domain, because the failure is the same
|
|
657
|
+
# whatever the agent does: the caller reads out something the agent then cannot find.
|
|
658
|
+
_QUOTED_VALUE = re.compile(
|
|
659
|
+
r"(?<![\w-])(?=[A-Za-z-]*\d)[A-Za-z0-9][A-Za-z0-9-]{3,}(?![\w-])"
|
|
660
|
+
)
|
|
661
|
+
|
|
662
|
+
# Values that look quotable but are never records the agent looks up.
|
|
663
|
+
_NOT_A_RECORD = re.compile(
|
|
664
|
+
r"^(?:\d{1,2}[:.]\d{2}|\d{1,4}(?:st|nd|rd|th)|20\d{2}|1?\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)$",
|
|
665
|
+
re.IGNORECASE,
|
|
666
|
+
)
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _quotable_values(text: str) -> set[str]:
|
|
670
|
+
"""Tokens in a piece of text that read as a value somebody would be asked to repeat."""
|
|
671
|
+
return {
|
|
672
|
+
token
|
|
673
|
+
for token in _QUOTED_VALUE.findall(text or "")
|
|
674
|
+
if not _NOT_A_RECORD.match(token)
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
# A value only has to be reachable if the caller is going to be asked for it. An address they are
|
|
679
|
+
# travelling to, or a price they are quoted, is the agent's to produce; a value they are told to say
|
|
680
|
+
# back is one the agent will check. Domain-neutral: the cue is the verb, not the kind of value.
|
|
681
|
+
_HANDED_OVER = re.compile(
|
|
682
|
+
r"(?:say|give|read|quote|provide|confirm|tell|repeat|use|enter|supply)\b[^.\n]{0,70}?"
|
|
683
|
+
r"(?<![\w-])((?=[A-Za-z-]*\d)[A-Za-z0-9][A-Za-z0-9-]{3,})(?![\w-])",
|
|
684
|
+
re.IGNORECASE,
|
|
685
|
+
)
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def _handed_to_caller(text: str) -> set[str]:
|
|
689
|
+
"""Values the instruction tells the caller to say back, which the agent will then check."""
|
|
690
|
+
return {
|
|
691
|
+
match.group(1)
|
|
692
|
+
for match in _HANDED_OVER.finditer(text or "")
|
|
693
|
+
if not _NOT_A_RECORD.match(match.group(1))
|
|
694
|
+
}
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
def naming_problems(scenario: Scenario) -> list[str]:
|
|
698
|
+
"""Whether the name says what is tested, or only who the agent was dealing with.
|
|
699
|
+
|
|
700
|
+
The folder name is how a failure is read weeks later. A caller's name in it says the caller was
|
|
701
|
+
carrying the difference the test should have been carrying, which is the same mistake as planning
|
|
702
|
+
a second scenario because the person could be somebody else. Measured on an earlier suite: twelve
|
|
703
|
+
of thirty one were still named for the caller after the skill asked them not to be, which is why
|
|
704
|
+
this is checked rather than requested.
|
|
705
|
+
"""
|
|
706
|
+
caller = str(getattr(scenario.persona, "name", "") or "").strip().lower()
|
|
707
|
+
if not caller:
|
|
708
|
+
return []
|
|
709
|
+
# Each part of the name, not the whole string: "marcus vance" is never a token of
|
|
710
|
+
# `refuse_expired_card_marcus`, so matching the full name lets every first-name suffix through.
|
|
711
|
+
parts = {part for part in caller.split() if len(part) > 2}
|
|
712
|
+
written = set(scenario.name.lower().replace("-", " ").replace("_", " ").split())
|
|
713
|
+
named_in = sorted(parts & written)
|
|
714
|
+
if not named_in:
|
|
715
|
+
return []
|
|
716
|
+
return [
|
|
717
|
+
f"the name contains the person's own name ({', '.join(named_in)}). Name it for the behaviour "
|
|
718
|
+
"under test, so a red result says which rule broke rather than who the agent was dealing "
|
|
719
|
+
"with, and so the suite sorts by what it covers rather than by who appeared in it"
|
|
720
|
+
]
|
|
721
|
+
|
|
722
|
+
|
|
723
|
+
def hollow_scenario_problems(scenario: Scenario) -> list[str]:
|
|
724
|
+
"""Whether the scenario tests reaching an outcome, or only the outcome itself.
|
|
725
|
+
|
|
726
|
+
A reference solution of one call, graded by one sub-goal naming that same call, is passed by an
|
|
727
|
+
agent that makes that call the moment it answers, having established nothing. Measured on a
|
|
728
|
+
suite of a hundred, eleven scenarios were a single `transfer_to_human` step graded by a single
|
|
729
|
+
`transferred_to_human` sub-goal, differing from each other only in the pretext, and every one of
|
|
730
|
+
them was passed by an agent that transfers every caller on arrival.
|
|
731
|
+
|
|
732
|
+
The bar is in the write skill and was not enough on its own, so it is checked here.
|
|
733
|
+
"""
|
|
734
|
+
if len(scenario.solution) > 1:
|
|
735
|
+
return []
|
|
736
|
+
if not scenario.solution:
|
|
737
|
+
return []
|
|
738
|
+
return [
|
|
739
|
+
"the reference solution is a single call and there is nothing the agent has to establish "
|
|
740
|
+
"first, so an agent that makes that call on arrival passes without doing any of the work. "
|
|
741
|
+
"Either the solution shows how the outcome is reached, gathering what the decision depends "
|
|
742
|
+
"on before making it, or this is not a scenario"
|
|
743
|
+
]
|
|
744
|
+
|
|
745
|
+
|
|
746
|
+
def alignment_problems(
|
|
747
|
+
scenario: Scenario, world_state: dict[str, list[dict[str, Any]]] | None = None
|
|
748
|
+
) -> list[str]:
|
|
749
|
+
"""Whether the values the caller is told are values the world actually holds.
|
|
750
|
+
|
|
751
|
+
The failure this exists for, seen across a whole suite: an instruction telling the caller a
|
|
752
|
+
verification code, a reference or an account number that the scenario never seeds and the world
|
|
753
|
+
never had. The call cannot succeed however well the agent behaves, and the result is reported as
|
|
754
|
+
a finding about the agent when it is a finding about the scenario.
|
|
755
|
+
|
|
756
|
+
Deliberately domain-neutral. A code, a booking reference, a policy number and an order id all
|
|
757
|
+
fail the same way, so the rule is about values rather than about any one kind of value: anything
|
|
758
|
+
the instruction hands the caller has to be somewhere the agent can reach, which means this
|
|
759
|
+
scenario's `setup_code` or the world it starts from. A fixture entry is not enough, because a
|
|
760
|
+
fixture describes what a scenario relies on and only `setup_code` changes what is there.
|
|
761
|
+
"""
|
|
762
|
+
told = _handed_to_caller(scenario.instruction)
|
|
763
|
+
if not told:
|
|
764
|
+
return []
|
|
765
|
+
reachable = _quotable_values(scenario.setup_code or "")
|
|
766
|
+
for step in scenario.solution:
|
|
767
|
+
reachable |= _quotable_values(json.dumps(step.arguments, default=str))
|
|
768
|
+
reachable |= _quotable_values(
|
|
769
|
+
json.dumps(step.environment_arguments, default=str)
|
|
770
|
+
)
|
|
771
|
+
if world_state:
|
|
772
|
+
reachable |= _quotable_values(json.dumps(world_state, default=str)[:200000])
|
|
773
|
+
missing = sorted(told - reachable)
|
|
774
|
+
if not missing:
|
|
775
|
+
return []
|
|
776
|
+
return [
|
|
777
|
+
"the instruction gives the caller "
|
|
778
|
+
+ ", ".join(missing)
|
|
779
|
+
+ " to say back, and neither setup_code nor the world holds "
|
|
780
|
+
+ ("them" if len(missing) > 1 else "it")
|
|
781
|
+
+ ". Seed what the caller is told, or tell them what is seeded. Naming a value in fixture "
|
|
782
|
+
"only declares it: setup_code is what the world ends up holding"
|
|
783
|
+
]
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
# What a setup does to the world, told apart by which call it makes. `put` adds a record and
|
|
787
|
+
# `call` drives a tool that produces one; `change` and `drop` only touch what was already there.
|
|
788
|
+
_CREATES_A_RECORD = re.compile(r"world\.(?:put|call)\s*\(")
|
|
789
|
+
_ONLY_TOUCHES_EXISTING = re.compile(r"world\.(?:change|drop)\s*\(")
|
|
790
|
+
|
|
791
|
+
|
|
792
|
+
def self_sufficiency_problems(scenario: Scenario) -> list[str]:
|
|
793
|
+
"""Whether this scenario owns the records its outcome turns on, or borrows them.
|
|
794
|
+
|
|
795
|
+
A setup that only adjusts rows it did not create is building the test on state it does not
|
|
796
|
+
control: the row belongs to the frozen base, so a second scenario adjusting the same row is
|
|
797
|
+
testing the same record from two directions and neither describes a world it owns. Measured on
|
|
798
|
+
a fan-out suite of 86, sixty seven were one or two `world.change` calls against base rows, four
|
|
799
|
+
scenarios deep on the same rider, and the reused verification codes were the visible symptom of
|
|
800
|
+
it.
|
|
801
|
+
|
|
802
|
+
An empty setup stays legal. That is the documented case where the target's store is
|
|
803
|
+
process-local with no seam, so the scenario cannot alter it and says so by touching nothing.
|
|
804
|
+
"""
|
|
805
|
+
body = (scenario.setup_code or "").strip()
|
|
806
|
+
if not body:
|
|
807
|
+
return []
|
|
808
|
+
if _CREATES_A_RECORD.search(body):
|
|
809
|
+
return []
|
|
810
|
+
if not _ONLY_TOUCHES_EXISTING.search(body):
|
|
811
|
+
return []
|
|
812
|
+
return [
|
|
813
|
+
"setup_code only adjusts records that were already there and creates none of its own, so "
|
|
814
|
+
"this scenario shares its data with every other scenario that touches the same records. "
|
|
815
|
+
"Create what the outcome turns on: its own person, its own record, its own code, with "
|
|
816
|
+
"world.put or by driving the agent's own tool. Shared reference data a whole world sits on "
|
|
817
|
+
"can be read as it is, but the thing being tested has to belong to this scenario"
|
|
818
|
+
]
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def _predictable(code: str) -> bool:
|
|
822
|
+
"""Whether a one-time code is one nobody would be issued.
|
|
823
|
+
|
|
824
|
+
The hand-kept list of obvious ones caught `111111` and `123456` and let `000111` through, which then
|
|
825
|
+
turned up twice in a 200-scenario suite. Tested as a property instead: a code built from one or two
|
|
826
|
+
digits, or one that simply counts up or down, is a placeholder however it is arranged.
|
|
827
|
+
"""
|
|
828
|
+
if not code.isdigit() or len(code) < 4:
|
|
829
|
+
return code in _WEAK_CODES
|
|
830
|
+
if len(set(code)) <= 2:
|
|
831
|
+
return True
|
|
832
|
+
steps = {ord(later) - ord(earlier) for earlier, later in zip(code, code[1:])}
|
|
833
|
+
if steps in ({1}, {-1}):
|
|
834
|
+
return True
|
|
835
|
+
return code in _WEAK_CODES
|
|
836
|
+
|
|
837
|
+
|
|
838
|
+
def fixture_problems(scenario: Scenario) -> list[str]:
|
|
839
|
+
"""Reject demo-shaped data before a paid run makes it look like production traffic."""
|
|
840
|
+
problems: list[str] = []
|
|
841
|
+
codes = _six_digit_values(scenario)
|
|
842
|
+
weak = sorted({code for code in codes if _predictable(code)})
|
|
843
|
+
if weak:
|
|
844
|
+
problems.append(
|
|
845
|
+
"fixture uses predictable verification code(s): "
|
|
846
|
+
+ ", ".join(weak)
|
|
847
|
+
+ ". Generate a different non-sequential six-digit value for this scenario"
|
|
848
|
+
)
|
|
849
|
+
written = json.dumps(
|
|
850
|
+
{
|
|
851
|
+
"instruction": scenario.instruction,
|
|
852
|
+
"persona": scenario.persona.model_dump() if scenario.persona else {},
|
|
853
|
+
"fixture": scenario.fixture,
|
|
854
|
+
"setup": scenario.setup_code,
|
|
855
|
+
},
|
|
856
|
+
default=str,
|
|
857
|
+
).lower()
|
|
858
|
+
clichés = [
|
|
859
|
+
value
|
|
860
|
+
for value in ("test user", "john doe", "jane doe", "123 main street")
|
|
861
|
+
if value in written
|
|
862
|
+
]
|
|
863
|
+
if clichés:
|
|
864
|
+
problems.append("fixture contains placeholder demo data: " + ", ".join(clichés))
|
|
865
|
+
card_endings = sorted(
|
|
866
|
+
set(
|
|
867
|
+
re.findall(
|
|
868
|
+
r"(?:last4|card_last4|payment_last4)[^\n]{0,30}?[\"']?(0000|1111|1234|4242|4444)[\"']?",
|
|
869
|
+
written,
|
|
870
|
+
)
|
|
871
|
+
)
|
|
872
|
+
)
|
|
873
|
+
if card_endings:
|
|
874
|
+
problems.append(
|
|
875
|
+
"fixture uses placeholder payment-card ending(s): "
|
|
876
|
+
+ ", ".join(card_endings)
|
|
877
|
+
)
|
|
878
|
+
spoken_card_endings = sorted(
|
|
879
|
+
set(
|
|
880
|
+
re.findall(
|
|
881
|
+
r"(?:ending(?:\s+in)?|last\s+four(?:\s+digits)?(?:\s+are)?)\D{0,12}"
|
|
882
|
+
r"(0000|1111|1234|4242|4444)",
|
|
883
|
+
written,
|
|
884
|
+
)
|
|
885
|
+
)
|
|
886
|
+
)
|
|
887
|
+
if spoken_card_endings:
|
|
888
|
+
problems.append(
|
|
889
|
+
"fixture/instruction uses placeholder payment-card ending(s): "
|
|
890
|
+
+ ", ".join(spoken_card_endings)
|
|
891
|
+
)
|
|
892
|
+
demo_ids = sorted(
|
|
893
|
+
value
|
|
894
|
+
for value in ("ub12345678", "booking123", "booking_123", "test123")
|
|
895
|
+
if value in written
|
|
896
|
+
)
|
|
897
|
+
demo_ids.extend(
|
|
898
|
+
re.findall(r"\b(?:ub_[a-z]+_0*1|pay_[a-z]+(?:_[a-z]+)*0*1)\b", written)
|
|
899
|
+
)
|
|
900
|
+
demo_ids = sorted(set(demo_ids))
|
|
901
|
+
if demo_ids:
|
|
902
|
+
problems.append(
|
|
903
|
+
"fixture uses placeholder transaction identifier(s): " + ", ".join(demo_ids)
|
|
904
|
+
)
|
|
905
|
+
return problems
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def rare_event_ceiling(suite_size: int) -> int:
|
|
909
|
+
"""The most scenarios of this suite size that may carry a rare call condition, rounded up."""
|
|
910
|
+
return max(1, ceil(suite_size * RARE_CONDITION_SHARE))
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
def suite_diversity_problems(scenarios: list[Scenario]) -> list[str]:
|
|
914
|
+
"""Whether a conversational suite represents meaningfully different people and data."""
|
|
915
|
+
if len(scenarios) < 4:
|
|
916
|
+
return []
|
|
917
|
+
problems: list[str] = []
|
|
918
|
+
personas = [one.persona for one in scenarios if one.persona]
|
|
919
|
+
names = [one.name.strip().lower() for one in personas if one and one.name.strip()]
|
|
920
|
+
unique_names = len(set(names))
|
|
921
|
+
required_names = min(len(scenarios), max(3, ceil(len(scenarios) * 0.9)))
|
|
922
|
+
if unique_names < required_names:
|
|
923
|
+
repeated = [name for name, count in Counter(names).items() if count > 2]
|
|
924
|
+
problems.append(
|
|
925
|
+
f"only {unique_names} distinct caller names across {len(scenarios)} scenarios; "
|
|
926
|
+
f"need at least {required_names}"
|
|
927
|
+
+ (f". Overused: {', '.join(repeated)}" if repeated else "")
|
|
928
|
+
)
|
|
929
|
+
openings = [
|
|
930
|
+
one.initial_message.strip().lower()
|
|
931
|
+
for one in personas
|
|
932
|
+
if one and one.initial_message.strip()
|
|
933
|
+
]
|
|
934
|
+
if len(set(openings)) != len(openings):
|
|
935
|
+
problems.append("caller opening messages repeat verbatim across scenarios")
|
|
936
|
+
locations = {
|
|
937
|
+
one.location.strip().lower() for one in personas if one and one.location.strip()
|
|
938
|
+
}
|
|
939
|
+
if len(scenarios) >= 8 and len(locations) < 3:
|
|
940
|
+
problems.append(
|
|
941
|
+
f"only {len(locations)} persona locations across {len(scenarios)} scenarios; need 3"
|
|
942
|
+
)
|
|
943
|
+
# An outbound suite that is all one awareness tests one opening repeatedly. Enforced rather than
|
|
944
|
+
# asked for: told to prefer `unaware`, writers made it the default and produced seven of eight,
|
|
945
|
+
# and told to cover more than one they had settled on `expecting` instead. Both leave two thirds
|
|
946
|
+
# of the opening untested.
|
|
947
|
+
outbound = [one for one in scenarios if one.call_direction == "outbound"]
|
|
948
|
+
if len(outbound) >= 3:
|
|
949
|
+
spread = Counter(one.caller_awareness or LEAST_AWARE for one in outbound)
|
|
950
|
+
if len(spread) < 2:
|
|
951
|
+
problems.append(
|
|
952
|
+
f"all {len(outbound)} outbound scenarios are caller_awareness "
|
|
953
|
+
f"{next(iter(spread))!r}; cover at least two of "
|
|
954
|
+
+ ", ".join(CALLER_AWARENESS)
|
|
955
|
+
)
|
|
956
|
+
elif max(spread.values()) > ceil(len(outbound) * 0.7):
|
|
957
|
+
worst, count = spread.most_common(1)[0]
|
|
958
|
+
problems.append(
|
|
959
|
+
f"{count} of {len(outbound)} outbound scenarios are caller_awareness {worst!r}; "
|
|
960
|
+
"keep any one of "
|
|
961
|
+
+ ", ".join(CALLER_AWARENESS)
|
|
962
|
+
+ " under 70 percent of them"
|
|
963
|
+
)
|
|
964
|
+
if not spread.get(LEAST_AWARE):
|
|
965
|
+
problems.append(
|
|
966
|
+
f"no outbound scenario has caller_awareness {LEAST_AWARE!r}, the one that tests whether "
|
|
967
|
+
"the agent says who it is and why it called before asking for anything"
|
|
968
|
+
)
|
|
969
|
+
# A mailbox tests one narrow thing, so it is worth a few scenarios and never a theme.
|
|
970
|
+
mailboxes = [one for one in scenarios if one.answered_by == VOICEMAIL]
|
|
971
|
+
allowed = rare_event_ceiling(len(scenarios))
|
|
972
|
+
if len(mailboxes) > allowed:
|
|
973
|
+
problems.append(
|
|
974
|
+
f"{len(mailboxes)} of {len(scenarios)} scenarios are answered_by {VOICEMAIL!r}; keep "
|
|
975
|
+
f"them to at most {allowed} here"
|
|
976
|
+
)
|
|
977
|
+
# Two or more have to be different mailboxes. Three would need forty one scenarios at this share.
|
|
978
|
+
if len(mailboxes) >= 2:
|
|
979
|
+
greetings = {
|
|
980
|
+
" ".join(
|
|
981
|
+
(one.persona.initial_message if one.persona else "").lower().split()
|
|
982
|
+
)
|
|
983
|
+
for one in mailboxes
|
|
984
|
+
}
|
|
985
|
+
if len(greetings - {""}) < 2:
|
|
986
|
+
problems.append(
|
|
987
|
+
f"all {len(mailboxes)} voicemail scenarios use the same greeting; vary it, since a "
|
|
988
|
+
"named personal mailbox, a carrier mailbox with no name, a full mailbox and a long "
|
|
989
|
+
"greeting are four different tests of the agent"
|
|
990
|
+
)
|
|
991
|
+
# Style is the stronger axis: it also decides whether a tone follows the greeting.
|
|
992
|
+
styles = {
|
|
993
|
+
str(one.voicemail_style or DEFAULT_VOICEMAIL_STYLE).strip().lower()
|
|
994
|
+
for one in mailboxes
|
|
995
|
+
}
|
|
996
|
+
if len(styles) < 2:
|
|
997
|
+
problems.append(
|
|
998
|
+
f"all {len(mailboxes)} voicemail scenarios are voicemail_style "
|
|
999
|
+
f"{next(iter(styles))!r}; cover at least two of "
|
|
1000
|
+
+ ", ".join(VOICEMAIL_STYLES)
|
|
1001
|
+
)
|
|
1002
|
+
# A code naturally appears several times inside one scenario (fixture, caller script,
|
|
1003
|
+
# reference verify call). Diversity is about reuse *between* callers, not repeated mention
|
|
1004
|
+
# of the same fact inside one test.
|
|
1005
|
+
codes = [
|
|
1006
|
+
code for scenario in scenarios for code in set(_six_digit_values(scenario))
|
|
1007
|
+
]
|
|
1008
|
+
duplicated_codes = sorted(
|
|
1009
|
+
code for code, count in Counter(codes).items() if count > 1
|
|
1010
|
+
)
|
|
1011
|
+
if duplicated_codes:
|
|
1012
|
+
problems.append(
|
|
1013
|
+
"verification codes are reused across scenarios: "
|
|
1014
|
+
+ ", ".join(duplicated_codes)
|
|
1015
|
+
)
|
|
1016
|
+
setups = [signature for one in scenarios if (signature := _setup_signature(one))]
|
|
1017
|
+
if len(set(setups)) != len(setups):
|
|
1018
|
+
problems.append("identical scenario setup data is reused more than once")
|
|
1019
|
+
return problems
|
|
1020
|
+
|
|
1021
|
+
|
|
1022
|
+
def _setup_signature(scenario: Scenario) -> str:
|
|
1023
|
+
"""Comparable setup code, excluding the generated no-op function/documentation."""
|
|
1024
|
+
source = scenario.setup_code.strip()
|
|
1025
|
+
if not source:
|
|
1026
|
+
return ""
|
|
1027
|
+
try:
|
|
1028
|
+
tree = ast.parse(source)
|
|
1029
|
+
except SyntaxError:
|
|
1030
|
+
return " ".join(source.split())
|
|
1031
|
+
function = next(
|
|
1032
|
+
(node for node in tree.body if isinstance(node, ast.FunctionDef)), None
|
|
1033
|
+
)
|
|
1034
|
+
if function is None:
|
|
1035
|
+
return " ".join(source.split())
|
|
1036
|
+
meaningful = [
|
|
1037
|
+
node
|
|
1038
|
+
for node in function.body
|
|
1039
|
+
if not isinstance(node, ast.Pass)
|
|
1040
|
+
and not (
|
|
1041
|
+
isinstance(node, ast.Expr)
|
|
1042
|
+
and isinstance(node.value, ast.Constant)
|
|
1043
|
+
and isinstance(node.value.value, str)
|
|
1044
|
+
)
|
|
1045
|
+
]
|
|
1046
|
+
return (
|
|
1047
|
+
"" if not meaningful else ast.dump(ast.Module(body=meaningful, type_ignores=[]))
|
|
1048
|
+
)
|