agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
"""One ALK harness executor used by the local CLI and hosted sandbox workers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import asyncio
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
import re
|
|
12
|
+
import subprocess
|
|
13
|
+
from typing import Callable, Protocol
|
|
14
|
+
|
|
15
|
+
from .job import (
|
|
16
|
+
FailureDomain,
|
|
17
|
+
HarnessFailure,
|
|
18
|
+
HarnessJob,
|
|
19
|
+
HarnessJobStatus,
|
|
20
|
+
HarnessStage,
|
|
21
|
+
SourceKind,
|
|
22
|
+
SourceVisibility,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class SourceAcquirer(Protocol):
|
|
27
|
+
"""Materialize job source inside an executor-owned ephemeral workspace."""
|
|
28
|
+
|
|
29
|
+
async def acquire(self, job: HarnessJob, workspace: Path) -> Path: ...
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class SourceAcquisitionError(RuntimeError):
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class GitHubSourceAcquirer:
|
|
37
|
+
"""Clone one platform-authorized repository without exposing its token in argv or logs."""
|
|
38
|
+
|
|
39
|
+
_REPOSITORY = re.compile(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$")
|
|
40
|
+
|
|
41
|
+
def __init__(self, installation_token: Callable[[str], str]) -> None:
|
|
42
|
+
self._installation_token = installation_token
|
|
43
|
+
|
|
44
|
+
async def acquire(self, job: HarnessJob, workspace: Path) -> Path:
|
|
45
|
+
source = job.source
|
|
46
|
+
repository = str(source.repository or "")
|
|
47
|
+
installation_id = str(source.installation_id or "")
|
|
48
|
+
if not self._REPOSITORY.fullmatch(repository):
|
|
49
|
+
raise SourceAcquisitionError("github_repository_invalid")
|
|
50
|
+
is_public = source.visibility is SourceVisibility.PUBLIC
|
|
51
|
+
if not is_public and not installation_id:
|
|
52
|
+
raise SourceAcquisitionError("github_installation_missing")
|
|
53
|
+
token = "" if is_public else self._installation_token(installation_id)
|
|
54
|
+
if not is_public and not token:
|
|
55
|
+
raise SourceAcquisitionError("github_installation_token_missing")
|
|
56
|
+
if source.ref and (
|
|
57
|
+
".." in source.ref or not re.fullmatch(r"[A-Za-z0-9._/-]+", source.ref)
|
|
58
|
+
):
|
|
59
|
+
raise SourceAcquisitionError("github_ref_invalid")
|
|
60
|
+
workspace = workspace.expanduser().resolve()
|
|
61
|
+
workspace.mkdir(parents=True, exist_ok=True)
|
|
62
|
+
destination = workspace / "repository"
|
|
63
|
+
if destination.exists():
|
|
64
|
+
raise SourceAcquisitionError(f"source_destination_not_empty: {destination}")
|
|
65
|
+
|
|
66
|
+
command = ["git", "clone", "--depth", "1"]
|
|
67
|
+
if source.ref:
|
|
68
|
+
command.extend(["--branch", source.ref])
|
|
69
|
+
command.extend([f"https://github.com/{repository}.git", str(destination)])
|
|
70
|
+
# Git reads the authorization header from its child environment. It never appears in
|
|
71
|
+
# the process command, exception, persisted job, or event stream.
|
|
72
|
+
environment = {**os.environ, "GIT_TERMINAL_PROMPT": "0"}
|
|
73
|
+
if token:
|
|
74
|
+
environment.update(
|
|
75
|
+
{
|
|
76
|
+
"GIT_CONFIG_COUNT": "1",
|
|
77
|
+
"GIT_CONFIG_KEY_0": "http.extraHeader",
|
|
78
|
+
"GIT_CONFIG_VALUE_0": f"Authorization: Bearer {token}",
|
|
79
|
+
}
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
def clone() -> None:
|
|
83
|
+
try:
|
|
84
|
+
completed = subprocess.run(
|
|
85
|
+
command,
|
|
86
|
+
env=environment,
|
|
87
|
+
cwd=workspace,
|
|
88
|
+
capture_output=True,
|
|
89
|
+
text=True,
|
|
90
|
+
timeout=300,
|
|
91
|
+
check=False,
|
|
92
|
+
)
|
|
93
|
+
except (OSError, subprocess.TimeoutExpired) as exc:
|
|
94
|
+
raise SourceAcquisitionError(
|
|
95
|
+
f"github_clone_unavailable: {type(exc).__name__}"
|
|
96
|
+
) from exc
|
|
97
|
+
if completed.returncode:
|
|
98
|
+
detail = (
|
|
99
|
+
completed.stderr or completed.stdout or "clone failed"
|
|
100
|
+
).strip()
|
|
101
|
+
# Git errors should not contain an env-only header, but redact defensively.
|
|
102
|
+
detail = detail.replace(token, "[REDACTED]")[:1000]
|
|
103
|
+
raise SourceAcquisitionError(f"github_clone_failed: {detail}")
|
|
104
|
+
|
|
105
|
+
if source.commit_sha:
|
|
106
|
+
verified = subprocess.run(
|
|
107
|
+
["git", "rev-parse", "HEAD"],
|
|
108
|
+
cwd=destination,
|
|
109
|
+
capture_output=True,
|
|
110
|
+
text=True,
|
|
111
|
+
timeout=30,
|
|
112
|
+
check=False,
|
|
113
|
+
env={**os.environ, "GIT_TERMINAL_PROMPT": "0"},
|
|
114
|
+
)
|
|
115
|
+
actual = verified.stdout.strip().lower()
|
|
116
|
+
if verified.returncode or actual != source.commit_sha.lower():
|
|
117
|
+
raise SourceAcquisitionError("github_commit_mismatch")
|
|
118
|
+
|
|
119
|
+
await asyncio.to_thread(clone)
|
|
120
|
+
if not destination.is_dir():
|
|
121
|
+
raise SourceAcquisitionError("github_clone_missing_checkout")
|
|
122
|
+
return destination
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class HarnessExecutor:
|
|
126
|
+
"""Execute the autonomous pipeline without platform or scheduler dependencies.
|
|
127
|
+
|
|
128
|
+
A hosted worker first resolves a GitHub installation/archive/image through its
|
|
129
|
+
``SourceAcquirer``. A local invocation already has a path. From that point onward both call
|
|
130
|
+
exactly the same ``auto`` pipeline and produce the same job, bundle, event, scenario, trace,
|
|
131
|
+
and result artifacts.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
async def run(
|
|
135
|
+
self,
|
|
136
|
+
job: HarnessJob,
|
|
137
|
+
*,
|
|
138
|
+
source: Path,
|
|
139
|
+
output: Path,
|
|
140
|
+
model: str | None = None,
|
|
141
|
+
run_model: str | None = None,
|
|
142
|
+
adjustments_path: Path | None = None,
|
|
143
|
+
) -> HarnessJobStatus:
|
|
144
|
+
from .cli import _auto
|
|
145
|
+
|
|
146
|
+
output = output.expanduser().resolve()
|
|
147
|
+
source = source.expanduser().resolve()
|
|
148
|
+
# A branch name is acquisition input, not immutable provenance. Resolve the checkout to
|
|
149
|
+
# its exact commit before job.json and the environment plan are written. The source
|
|
150
|
+
# content digest remains authoritative for uploads and non-Git sources.
|
|
151
|
+
if job.source.kind is SourceKind.GITHUB and not job.source.commit_sha:
|
|
152
|
+
completed = await asyncio.to_thread(
|
|
153
|
+
subprocess.run,
|
|
154
|
+
["git", "rev-parse", "HEAD"],
|
|
155
|
+
cwd=source,
|
|
156
|
+
capture_output=True,
|
|
157
|
+
text=True,
|
|
158
|
+
timeout=30,
|
|
159
|
+
check=False,
|
|
160
|
+
env={**os.environ, "GIT_TERMINAL_PROMPT": "0"},
|
|
161
|
+
)
|
|
162
|
+
commit_sha = completed.stdout.strip().lower()
|
|
163
|
+
if completed.returncode or not re.fullmatch(r"[0-9a-f]{40}", commit_sha):
|
|
164
|
+
return HarnessJobStatus(
|
|
165
|
+
job_id=job.job_id,
|
|
166
|
+
run_id=job.run_id,
|
|
167
|
+
stage=HarnessStage.FAILED,
|
|
168
|
+
updated_at=datetime.now(timezone.utc),
|
|
169
|
+
detail="could not resolve the acquired GitHub revision",
|
|
170
|
+
failure=HarnessFailure(
|
|
171
|
+
domain=FailureDomain.INFRASTRUCTURE,
|
|
172
|
+
stage=HarnessStage.ACQUIRING_SOURCE,
|
|
173
|
+
code="github_commit_resolution_failed",
|
|
174
|
+
message="The runner could not resolve the acquired GitHub revision",
|
|
175
|
+
),
|
|
176
|
+
total_scenarios=job.scenario_count,
|
|
177
|
+
)
|
|
178
|
+
job = job.model_copy(
|
|
179
|
+
update={
|
|
180
|
+
"source": job.source.model_copy(update={"commit_sha": commit_sha})
|
|
181
|
+
}
|
|
182
|
+
)
|
|
183
|
+
# Source acquisition has already materialized GitHub/archive/local inputs as a local
|
|
184
|
+
# checkout. The understanding registry describes how to inspect that materialized
|
|
185
|
+
# content (``repo``), while the immutable job retains its original source provenance.
|
|
186
|
+
# Passing transport kinds such as ``archive`` into source resolution makes a valid
|
|
187
|
+
# uploaded repository fail before its code is inspected.
|
|
188
|
+
understanding_kind = str(job.metadata.get("source_kind") or "repo")
|
|
189
|
+
if understanding_kind not in {"repo", "spec"}:
|
|
190
|
+
understanding_kind = "repo"
|
|
191
|
+
args = argparse.Namespace(
|
|
192
|
+
path=str(source),
|
|
193
|
+
name=str(job.metadata.get("agent_name") or source.name),
|
|
194
|
+
kind=understanding_kind,
|
|
195
|
+
out=str(output),
|
|
196
|
+
count=job.scenario_count,
|
|
197
|
+
model=model,
|
|
198
|
+
run_model=run_model,
|
|
199
|
+
job=job,
|
|
200
|
+
adjustments_path=str(adjustments_path) if adjustments_path else None,
|
|
201
|
+
)
|
|
202
|
+
status = await _auto(args)
|
|
203
|
+
failure = _failure_from_events(output) if status not in (0, 2) else None
|
|
204
|
+
return HarnessJobStatus(
|
|
205
|
+
job_id=job.job_id,
|
|
206
|
+
run_id=job.run_id,
|
|
207
|
+
stage=HarnessStage.COMPLETED if status in (0, 2) else HarnessStage.FAILED,
|
|
208
|
+
updated_at=datetime.now(timezone.utc),
|
|
209
|
+
detail=(
|
|
210
|
+
"agent checks failed"
|
|
211
|
+
if status == 2
|
|
212
|
+
else None
|
|
213
|
+
if status == 0
|
|
214
|
+
else f"exit {status}"
|
|
215
|
+
),
|
|
216
|
+
failure=failure,
|
|
217
|
+
completed_scenarios=_scenario_count(output) if status in (0, 2) else 0,
|
|
218
|
+
total_scenarios=_scenario_count(output) or job.scenario_count,
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
async def acquire_and_run(
|
|
222
|
+
self,
|
|
223
|
+
job: HarnessJob,
|
|
224
|
+
*,
|
|
225
|
+
acquirer: SourceAcquirer,
|
|
226
|
+
workspace: Path,
|
|
227
|
+
output: Path,
|
|
228
|
+
model: str | None = None,
|
|
229
|
+
run_model: str | None = None,
|
|
230
|
+
) -> HarnessJobStatus:
|
|
231
|
+
source = await acquirer.acquire(job, workspace)
|
|
232
|
+
return await self.run(
|
|
233
|
+
job,
|
|
234
|
+
source=source,
|
|
235
|
+
output=output,
|
|
236
|
+
model=model,
|
|
237
|
+
run_model=run_model,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def run_sync(job: HarnessJob, *, source: Path, output: Path) -> HarnessJobStatus:
|
|
242
|
+
"""Small synchronous adapter for job consumers that do not own an event loop."""
|
|
243
|
+
return asyncio.run(HarnessExecutor().run(job, source=source, output=output))
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _scenario_count(output: Path) -> int:
|
|
247
|
+
try:
|
|
248
|
+
value = json.loads((output / "scenarios.json").read_text(encoding="utf-8"))
|
|
249
|
+
except (OSError, ValueError):
|
|
250
|
+
return 0
|
|
251
|
+
return len(value) if isinstance(value, list) else 0
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _failure_from_events(output: Path) -> HarnessFailure:
|
|
255
|
+
failed: dict = {}
|
|
256
|
+
path = output / "harness-events.jsonl"
|
|
257
|
+
if path.is_file():
|
|
258
|
+
for raw in path.read_text(encoding="utf-8", errors="replace").splitlines():
|
|
259
|
+
try:
|
|
260
|
+
event = json.loads(raw)
|
|
261
|
+
except ValueError:
|
|
262
|
+
continue
|
|
263
|
+
if event.get("type") == "harness.stage.failed":
|
|
264
|
+
failed = event.get("payload") or {}
|
|
265
|
+
label = str(failed.get("stage") or "running")
|
|
266
|
+
stage, domain = {
|
|
267
|
+
"understand": (HarnessStage.UNDERSTANDING_AGENT, FailureDomain.AGENT),
|
|
268
|
+
"environment": (
|
|
269
|
+
HarnessStage.VALIDATING_ENVIRONMENT,
|
|
270
|
+
FailureDomain.ENVIRONMENT,
|
|
271
|
+
),
|
|
272
|
+
"scenarios": (
|
|
273
|
+
HarnessStage.VALIDATING_SCENARIOS,
|
|
274
|
+
FailureDomain.SIMULATOR,
|
|
275
|
+
),
|
|
276
|
+
"calls": (HarnessStage.RUNNING, FailureDomain.CONNECTIVITY),
|
|
277
|
+
"cleaning_up": (
|
|
278
|
+
HarnessStage.CLEANING_UP,
|
|
279
|
+
FailureDomain.INFRASTRUCTURE,
|
|
280
|
+
),
|
|
281
|
+
"uploading_artifacts": (
|
|
282
|
+
HarnessStage.UPLOADING_ARTIFACTS,
|
|
283
|
+
FailureDomain.ARTIFACT,
|
|
284
|
+
),
|
|
285
|
+
}.get(label, (HarnessStage.RUNNING, FailureDomain.INFRASTRUCTURE))
|
|
286
|
+
return HarnessFailure(
|
|
287
|
+
domain=domain,
|
|
288
|
+
stage=stage,
|
|
289
|
+
code=str(failed.get("code") or f"{label}_failed"),
|
|
290
|
+
message=str(failed.get("detail") or f"Harness stage {label} failed"),
|
|
291
|
+
# A job replay can repeat real calls. Only a failing stage that explicitly proves
|
|
292
|
+
# it is safe may opt in to retry; deterministic agent/grading failures never do.
|
|
293
|
+
retryable=bool(failed.get("retryable", False)),
|
|
294
|
+
details={"status": failed.get("status", 1)},
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
__all__ = [
|
|
299
|
+
"GitHubSourceAcquirer",
|
|
300
|
+
"HarnessExecutor",
|
|
301
|
+
"SourceAcquirer",
|
|
302
|
+
"SourceAcquisitionError",
|
|
303
|
+
"run_sync",
|
|
304
|
+
]
|
fi/alk/harness/folder.py
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""A scenario as a folder of files, and running the code inside it.
|
|
2
|
+
|
|
3
|
+
A scenario used to be a row in one big JSON file, and its setup was a list of rows to insert.
|
|
4
|
+
That was enough while every world was a database. It stopped being enough the moment a world
|
|
5
|
+
could hold a service as well as a table: "the weather service starts returning errors" is not
|
|
6
|
+
expressible as rows, and neither is "the file is missing" or "the queue is backed up".
|
|
7
|
+
|
|
8
|
+
So a scenario owns a folder, and the parts that are logic are files:
|
|
9
|
+
|
|
10
|
+
scenarios/<name>/
|
|
11
|
+
scenario.json what it is: instruction, solution, which sub-goals
|
|
12
|
+
setup.py def setup(world) — the changes this scenario makes
|
|
13
|
+
ready.py def ready(world) — is the world ready for this scenario
|
|
14
|
+
checks/<goal>.py def check(world, calls) — one per deterministic sub-goal
|
|
15
|
+
|
|
16
|
+
The files are the artifact, not a rendering of one. Each is executable on its own, so a check
|
|
17
|
+
can be run by hand against what a run left behind and answer exactly what it answers inside the
|
|
18
|
+
harness. That is the whole point of them being files: something you can open, read and run is
|
|
19
|
+
something you can argue with.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from .catalogue import Catalogue
|
|
30
|
+
from .scenario import Scenario
|
|
31
|
+
from .world.runtime import GeneratedWorld
|
|
32
|
+
|
|
33
|
+
SCENARIOS = "scenarios"
|
|
34
|
+
INDEX = "scenarios.json"
|
|
35
|
+
|
|
36
|
+
# Appended to every check file the harness writes. The model writes only ``check(world, calls)``;
|
|
37
|
+
# this is what makes that same file runnable by a person, so nobody has to keep two versions of
|
|
38
|
+
# one truth in step.
|
|
39
|
+
_RUNNABLE = """
|
|
40
|
+
|
|
41
|
+
if __name__ == "__main__":
|
|
42
|
+
# Run this check by hand against what a run left behind. The first argument is anything
|
|
43
|
+
# inside the saved world's folder, because not every world has a database to name:
|
|
44
|
+
# python <this file> <world folder>/manifest.json [calls.json]
|
|
45
|
+
import json as _json
|
|
46
|
+
import sys as _sys
|
|
47
|
+
from pathlib import Path as _Path
|
|
48
|
+
|
|
49
|
+
from fi.alk.harness.world.runtime import Call as _Call
|
|
50
|
+
from fi.alk.harness.world.snapshot import restore as _restore
|
|
51
|
+
|
|
52
|
+
_world = _restore(_Path(_sys.argv[1]).parent) if len(_sys.argv) > 1 else None
|
|
53
|
+
_calls = []
|
|
54
|
+
if len(_sys.argv) > 2:
|
|
55
|
+
_calls = [_Call(**_one) for _one in _json.loads(_Path(_sys.argv[2]).read_text())]
|
|
56
|
+
_said = check(_world, _calls)
|
|
57
|
+
_held = _said is None or _said is True or (isinstance(_said, str) and not _said.strip())
|
|
58
|
+
print("held" if _held else f"FAILED: {_said}")
|
|
59
|
+
raise SystemExit(0 if _held else 1)
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Outcome:
|
|
65
|
+
"""What one piece of a scenario's own code did."""
|
|
66
|
+
|
|
67
|
+
ok: bool
|
|
68
|
+
said: str = ""
|
|
69
|
+
broken: bool = False
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _run(source: str, name: str, entry: str, *args: Any) -> Outcome:
|
|
73
|
+
"""Execute one function out of a scenario's own code.
|
|
74
|
+
|
|
75
|
+
A file that will not compile, or that raises, is **broken** rather than failing: it is our
|
|
76
|
+
mistake, and scoring it as though the world were wrong would send somebody looking in the
|
|
77
|
+
wrong place.
|
|
78
|
+
"""
|
|
79
|
+
if not source.strip():
|
|
80
|
+
return Outcome(True)
|
|
81
|
+
namespace: dict[str, Any] = {}
|
|
82
|
+
try:
|
|
83
|
+
exec(compile(source, f"<{name}>", "exec"), namespace)
|
|
84
|
+
except Exception as failed:
|
|
85
|
+
return Outcome(False, f"{name} would not compile: {failed}", broken=True)
|
|
86
|
+
|
|
87
|
+
function = namespace.get(entry)
|
|
88
|
+
if not callable(function):
|
|
89
|
+
return Outcome(False, f"{name} defines no {entry}()", broken=True)
|
|
90
|
+
try:
|
|
91
|
+
said = function(*args)
|
|
92
|
+
except Exception as failed:
|
|
93
|
+
return Outcome(
|
|
94
|
+
False, f"{name} raised {type(failed).__name__}: {failed}", broken=True
|
|
95
|
+
)
|
|
96
|
+
# The convention is that a complaint is a sentence, and anything else means it held. An empty
|
|
97
|
+
# string is the case worth naming: it reads as "no complaint" to whoever wrote it, and taking
|
|
98
|
+
# it as a failure produces a rejection with no reason attached, which cannot be acted on and
|
|
99
|
+
# sends the author hunting for a problem that is not there.
|
|
100
|
+
if said is None or said is True or (isinstance(said, str) and not said.strip()):
|
|
101
|
+
return Outcome(True)
|
|
102
|
+
if said is False:
|
|
103
|
+
# Bare False from ready() names no precondition, so a scenario that hits it cannot be
|
|
104
|
+
# told apart from one whose ready.py is simply wrong — that is our mistake, not a
|
|
105
|
+
# generation-time precondition failure, so ready() alone reports it broken.
|
|
106
|
+
return Outcome(
|
|
107
|
+
False,
|
|
108
|
+
f"{name} returned False without saying what is wrong. Return the sentence instead, "
|
|
109
|
+
"or None if it holds.",
|
|
110
|
+
broken=(entry == "ready"),
|
|
111
|
+
)
|
|
112
|
+
if entry == "ready" and not isinstance(said, str):
|
|
113
|
+
# Same reasoning as bare False, widened: ready() has no way to turn a non-string value
|
|
114
|
+
# into a precondition sentence, so any of them is our mistake rather than the world's.
|
|
115
|
+
return Outcome(
|
|
116
|
+
False,
|
|
117
|
+
f"{name} returned {type(said).__name__} {repr(said)[:200]}. Return the sentence "
|
|
118
|
+
"naming what is missing, or None if it holds.",
|
|
119
|
+
broken=True,
|
|
120
|
+
)
|
|
121
|
+
return Outcome(False, str(said))
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def apply_setup(scenario: Scenario, world: GeneratedWorld) -> Outcome:
|
|
125
|
+
"""Make this scenario's changes to the world."""
|
|
126
|
+
return _run(scenario.setup_code, f"{scenario.name}/setup.py", "setup", world)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def check_ready(scenario: Scenario, world: GeneratedWorld) -> Outcome:
|
|
130
|
+
"""Whether the world now holds what this scenario presumes."""
|
|
131
|
+
return _run(scenario.ready_code, f"{scenario.name}/ready.py", "ready", world)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def folder_for(destination: Path, name: str) -> Path:
|
|
135
|
+
return Path(destination) / SCENARIOS / name
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def write_folder(scenario: Scenario, catalogue: Catalogue, destination: Path) -> Path:
|
|
139
|
+
"""Write one scenario out as its own folder of files."""
|
|
140
|
+
root = folder_for(destination, scenario.name)
|
|
141
|
+
(root / "checks").mkdir(parents=True, exist_ok=True)
|
|
142
|
+
|
|
143
|
+
body = scenario.model_dump()
|
|
144
|
+
# The code lives in its own files; keeping a second copy in the JSON would let the two drift
|
|
145
|
+
# and leave nobody able to say which one ran.
|
|
146
|
+
body.pop("setup_code", None)
|
|
147
|
+
body.pop("ready_code", None)
|
|
148
|
+
(root / "scenario.json").write_text(
|
|
149
|
+
json.dumps(body, indent=2, ensure_ascii=False), encoding="utf-8"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
(root / "setup.py").write_text(
|
|
153
|
+
scenario.setup_code
|
|
154
|
+
or 'def setup(world):\n """This scenario runs on the base world unchanged."""\n',
|
|
155
|
+
encoding="utf-8",
|
|
156
|
+
)
|
|
157
|
+
(root / "ready.py").write_text(
|
|
158
|
+
scenario.ready_code
|
|
159
|
+
or 'def ready(world):\n """Nothing beyond the base world is presumed."""\n',
|
|
160
|
+
encoding="utf-8",
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
for name in scenario.sub_goals:
|
|
164
|
+
sub_goal = catalogue.named(name)
|
|
165
|
+
if sub_goal is None or not sub_goal.deterministic():
|
|
166
|
+
continue
|
|
167
|
+
(root / "checks" / f"{name}.py").write_text(
|
|
168
|
+
sub_goal.check.rstrip() + "\n" + _RUNNABLE, encoding="utf-8"
|
|
169
|
+
)
|
|
170
|
+
return root
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def read_folder(destination: Path, name: str) -> Scenario | None:
|
|
174
|
+
"""One scenario, reassembled from its folder."""
|
|
175
|
+
root = folder_for(destination, name)
|
|
176
|
+
body = root / "scenario.json"
|
|
177
|
+
if not body.exists():
|
|
178
|
+
return None
|
|
179
|
+
payload = json.loads(body.read_text(encoding="utf-8"))
|
|
180
|
+
for field, filename in (("setup_code", "setup.py"), ("ready_code", "ready.py")):
|
|
181
|
+
path = root / filename
|
|
182
|
+
payload[field] = path.read_text(encoding="utf-8") if path.exists() else ""
|
|
183
|
+
return Scenario.model_validate(payload)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def write_index(scenarios: list[Scenario], destination: Path) -> Path:
|
|
187
|
+
"""The whole suite at a glance, over the folders.
|
|
188
|
+
|
|
189
|
+
Regenerated from the folders rather than maintained alongside them, so it can never disagree
|
|
190
|
+
with what is actually on disk.
|
|
191
|
+
"""
|
|
192
|
+
destination = Path(destination)
|
|
193
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
194
|
+
path = destination / INDEX
|
|
195
|
+
path.write_text(
|
|
196
|
+
json.dumps(
|
|
197
|
+
[
|
|
198
|
+
{
|
|
199
|
+
"name": one.name,
|
|
200
|
+
"use_case": one.use_case,
|
|
201
|
+
"tests": one.tests,
|
|
202
|
+
"instruction": one.instruction,
|
|
203
|
+
"sub_goals": one.sub_goals,
|
|
204
|
+
"steps": len(one.solution),
|
|
205
|
+
"folder": f"{SCENARIOS}/{one.name}",
|
|
206
|
+
}
|
|
207
|
+
for one in scenarios
|
|
208
|
+
],
|
|
209
|
+
indent=2,
|
|
210
|
+
ensure_ascii=False,
|
|
211
|
+
),
|
|
212
|
+
encoding="utf-8",
|
|
213
|
+
)
|
|
214
|
+
return path
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def read_all(destination: Path) -> list[Scenario]:
|
|
218
|
+
"""Every scenario on disk, read from the folders."""
|
|
219
|
+
root = Path(destination) / SCENARIOS
|
|
220
|
+
if not root.exists():
|
|
221
|
+
return []
|
|
222
|
+
found: list[Scenario] = []
|
|
223
|
+
for folder in sorted(root.iterdir()):
|
|
224
|
+
if not folder.is_dir():
|
|
225
|
+
continue
|
|
226
|
+
try:
|
|
227
|
+
scenario = read_folder(destination, folder.name)
|
|
228
|
+
except Exception:
|
|
229
|
+
# A folder we cannot read is skipped rather than crashing the stage: the rest of the
|
|
230
|
+
# suite is still usable, and the gap shows up as a missing scenario.
|
|
231
|
+
continue
|
|
232
|
+
if scenario is not None:
|
|
233
|
+
found.append(scenario)
|
|
234
|
+
return found
|