agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""Where an agent comes from, and how a session reaches it.
|
|
2
|
+
|
|
3
|
+
A folder of source code is one kind of agent, not the only kind. The same agent may arrive as a
|
|
4
|
+
provider connection with a system prompt and a tool schema, as a platform definition, or as a
|
|
5
|
+
spec somebody pasted in. The stage that reads an agent is the same in all of those cases; what
|
|
6
|
+
differs is where it looks and what it is allowed to touch.
|
|
7
|
+
|
|
8
|
+
So the method stays in the skill and the location lives here. Supporting a new kind of agent is
|
|
9
|
+
registering one class, not editing any stage.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import subprocess
|
|
16
|
+
from collections.abc import Callable
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any, Protocol
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class AgentSource(Protocol):
|
|
23
|
+
"""Everything a stage needs in order to reach one agent."""
|
|
24
|
+
|
|
25
|
+
kind: str
|
|
26
|
+
name: str
|
|
27
|
+
|
|
28
|
+
def workdir(self) -> Path:
|
|
29
|
+
"""The directory the session runs in."""
|
|
30
|
+
|
|
31
|
+
def builtin_tools(self) -> tuple[str, ...]:
|
|
32
|
+
"""Built-in tools this source needs granted."""
|
|
33
|
+
|
|
34
|
+
def servers(self) -> dict[str, Any]:
|
|
35
|
+
"""In-process tool servers this source provides, if any."""
|
|
36
|
+
|
|
37
|
+
def briefing(self) -> str:
|
|
38
|
+
"""What to tell the model about where this agent's truth lives."""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class RepoSource:
|
|
43
|
+
"""An agent that exists as source code on disk."""
|
|
44
|
+
|
|
45
|
+
name: str
|
|
46
|
+
root: Path
|
|
47
|
+
kind: str = "repo"
|
|
48
|
+
|
|
49
|
+
def workdir(self) -> Path:
|
|
50
|
+
return self.root
|
|
51
|
+
|
|
52
|
+
def builtin_tools(self) -> tuple[str, ...]:
|
|
53
|
+
return ("Read", "Glob", "Grep")
|
|
54
|
+
|
|
55
|
+
def servers(self) -> dict[str, Any]:
|
|
56
|
+
return {}
|
|
57
|
+
|
|
58
|
+
def briefing(self) -> str:
|
|
59
|
+
ignored = {
|
|
60
|
+
".git",
|
|
61
|
+
".venv",
|
|
62
|
+
"node_modules",
|
|
63
|
+
"__pycache__",
|
|
64
|
+
"artifacts",
|
|
65
|
+
"build",
|
|
66
|
+
"dist",
|
|
67
|
+
}
|
|
68
|
+
indexed: list[str] = []
|
|
69
|
+
for path in sorted(self.root.rglob("*")):
|
|
70
|
+
try:
|
|
71
|
+
relative = path.relative_to(self.root)
|
|
72
|
+
except ValueError:
|
|
73
|
+
continue
|
|
74
|
+
if any(part in ignored for part in relative.parts) or not path.is_file():
|
|
75
|
+
continue
|
|
76
|
+
indexed.append(relative.as_posix())
|
|
77
|
+
if len(indexed) >= 240:
|
|
78
|
+
break
|
|
79
|
+
return (
|
|
80
|
+
f"This agent is a repository at {self.root}. Its truth is the source code: the tool "
|
|
81
|
+
"registrations, the function signatures, the validation logic, and whatever holds "
|
|
82
|
+
"its data. Read it with Read, Glob and Grep. Documentation describes intent; the "
|
|
83
|
+
"code describes behaviour, and where they disagree the code wins. Never glob the "
|
|
84
|
+
"entire repository: dependency caches such as .venv and node_modules are irrelevant. "
|
|
85
|
+
"Start from this pre-indexed source/config file list and read only relevant files:\n"
|
|
86
|
+
+ "\n".join(f"- {name}" for name in indexed)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass
|
|
91
|
+
class GitHubSource(RepoSource):
|
|
92
|
+
"""A public GitHub repository cloned into this harness session."""
|
|
93
|
+
|
|
94
|
+
url: str = ""
|
|
95
|
+
kind: str = "github"
|
|
96
|
+
|
|
97
|
+
def briefing(self) -> str:
|
|
98
|
+
return (
|
|
99
|
+
f"This agent was cloned from {self.url or 'GitHub'} into {self.root}. Its truth is "
|
|
100
|
+
"the cloned source code: the tool registrations, function signatures, validation "
|
|
101
|
+
"logic, and whatever holds its data. Read it with Read, Glob and Grep."
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def clone_github_repository(url: str, destination: Path) -> Path:
|
|
106
|
+
"""Shallow-clone one public GitHub repository into a session-owned directory."""
|
|
107
|
+
from .github import parse_github_location
|
|
108
|
+
|
|
109
|
+
try:
|
|
110
|
+
location = parse_github_location(url)
|
|
111
|
+
except ValueError as exc:
|
|
112
|
+
raise ValueError(
|
|
113
|
+
"use a public HTTPS GitHub repository or branch URL, such as "
|
|
114
|
+
"https://github.com/owner/repo/tree/branch"
|
|
115
|
+
) from exc
|
|
116
|
+
if destination.exists():
|
|
117
|
+
raise ValueError(f"the session source directory already exists: {destination}")
|
|
118
|
+
|
|
119
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
120
|
+
command = ["git", "clone", "--depth", "1"]
|
|
121
|
+
if location.ref:
|
|
122
|
+
command.extend(["--branch", location.ref])
|
|
123
|
+
command.extend([location.clone_url, str(destination)])
|
|
124
|
+
completed = subprocess.run(
|
|
125
|
+
command,
|
|
126
|
+
capture_output=True,
|
|
127
|
+
check=False,
|
|
128
|
+
text=True,
|
|
129
|
+
)
|
|
130
|
+
if completed.returncode:
|
|
131
|
+
detail = completed.stderr.strip() or "git clone failed"
|
|
132
|
+
raise RuntimeError(detail)
|
|
133
|
+
return destination
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@dataclass
|
|
137
|
+
class SpecSource:
|
|
138
|
+
"""An agent supplied directly as a prompt and a tool schema, with no repository.
|
|
139
|
+
|
|
140
|
+
This is the shape a hosted provider gives back, so it is also the fallback whenever a
|
|
141
|
+
connection can be read once and handed over as text.
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
name: str
|
|
145
|
+
system_prompt: str
|
|
146
|
+
tool_schema: list[dict[str, Any]] = field(default_factory=list)
|
|
147
|
+
data: dict[str, Any] = field(default_factory=dict)
|
|
148
|
+
scratch: Path = Path(".")
|
|
149
|
+
kind: str = "spec"
|
|
150
|
+
|
|
151
|
+
def workdir(self) -> Path:
|
|
152
|
+
return self.scratch
|
|
153
|
+
|
|
154
|
+
def builtin_tools(self) -> tuple[str, ...]:
|
|
155
|
+
return ()
|
|
156
|
+
|
|
157
|
+
def servers(self) -> dict[str, Any]:
|
|
158
|
+
return {}
|
|
159
|
+
|
|
160
|
+
def briefing(self) -> str:
|
|
161
|
+
parts = [
|
|
162
|
+
"This agent is supplied as a definition, not a repository. Everything knowable "
|
|
163
|
+
"about it is below; there is no code to open, so do not guess at anything absent.",
|
|
164
|
+
f"SYSTEM PROMPT:\n{self.system_prompt}",
|
|
165
|
+
]
|
|
166
|
+
if self.tool_schema:
|
|
167
|
+
parts.append(
|
|
168
|
+
f"TOOL SCHEMA:\n{json.dumps(self.tool_schema, indent=2)[:6000]}"
|
|
169
|
+
)
|
|
170
|
+
if self.data:
|
|
171
|
+
parts.append(f"DATA:\n{json.dumps(self.data, indent=2)[:6000]}")
|
|
172
|
+
return "\n\n".join(parts)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@dataclass
|
|
176
|
+
class ProviderSource:
|
|
177
|
+
"""A sanitized definition fetched from an externally hosted provider.
|
|
178
|
+
|
|
179
|
+
A connect-only provider agent has no repository in the sandbox. Representing its empty
|
|
180
|
+
source directory as a :class:`RepoSource` gives an authoring model filesystem tools and can
|
|
181
|
+
make it wander outside that directory looking for an implementation. The provider profile
|
|
182
|
+
is the complete source of truth for this mode, so expose only that profile and no file tools.
|
|
183
|
+
"""
|
|
184
|
+
|
|
185
|
+
name: str
|
|
186
|
+
profile: dict[str, Any]
|
|
187
|
+
scratch: Path = Path(".")
|
|
188
|
+
kind: str = "provider"
|
|
189
|
+
|
|
190
|
+
def workdir(self) -> Path:
|
|
191
|
+
return self.scratch
|
|
192
|
+
|
|
193
|
+
def builtin_tools(self) -> tuple[str, ...]:
|
|
194
|
+
return ()
|
|
195
|
+
|
|
196
|
+
def servers(self) -> dict[str, Any]:
|
|
197
|
+
return {}
|
|
198
|
+
|
|
199
|
+
def briefing(self) -> str:
|
|
200
|
+
return (
|
|
201
|
+
"This is an externally hosted provider agent, not a repository. The sanitized "
|
|
202
|
+
"provider definition below is authoritative for its conversation, prompt, model, "
|
|
203
|
+
"voice, states, and tool schemas. There is no source code to search or open. Do not "
|
|
204
|
+
"invent behavior or tool inputs that are absent from this definition.\n\n"
|
|
205
|
+
f"PROVIDER DEFINITION:\n{json.dumps(self.profile, indent=2, sort_keys=True)}"
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
_REGISTRY: dict[str, Callable[..., AgentSource]] = {
|
|
210
|
+
"repo": lambda **kw: RepoSource(name=kw["name"], root=Path(kw["root"])),
|
|
211
|
+
"github": lambda **kw: GitHubSource(
|
|
212
|
+
name=kw["name"], root=Path(kw["root"]), url=kw.get("url", "")
|
|
213
|
+
),
|
|
214
|
+
"spec": lambda **kw: SpecSource(
|
|
215
|
+
name=kw["name"],
|
|
216
|
+
system_prompt=kw.get("system_prompt", ""),
|
|
217
|
+
tool_schema=kw.get("tool_schema") or [],
|
|
218
|
+
data=kw.get("data") or {},
|
|
219
|
+
scratch=Path(kw.get("scratch", ".")),
|
|
220
|
+
),
|
|
221
|
+
"provider": lambda **kw: ProviderSource(
|
|
222
|
+
name=kw["name"],
|
|
223
|
+
profile=kw.get("profile") or {},
|
|
224
|
+
scratch=Path(kw.get("scratch", ".")),
|
|
225
|
+
),
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def register_source(kind: str, factory: Callable[..., AgentSource]) -> None:
|
|
230
|
+
"""Add a kind of agent. A provider connection is a class and one line here."""
|
|
231
|
+
_REGISTRY[kind] = factory
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def resolve(kind: str, **kwargs: Any) -> AgentSource:
|
|
235
|
+
if kind not in _REGISTRY:
|
|
236
|
+
raise NotImplementedError(
|
|
237
|
+
f"no agent source of kind {kind!r}; registered kinds are "
|
|
238
|
+
f"{', '.join(sorted(_REGISTRY))}"
|
|
239
|
+
)
|
|
240
|
+
# An empty root used to resolve to the current directory, which is worse than failing: every
|
|
241
|
+
# later stage then reads a real path, finds the harness's own repository, and reports that the
|
|
242
|
+
# agent has no code on disk. Nothing downstream can tell that apart from an agent that really
|
|
243
|
+
# was given as a specification.
|
|
244
|
+
if "root" in kwargs and not str(kwargs.get("root") or "").strip():
|
|
245
|
+
raise ValueError(
|
|
246
|
+
f"a {kind!r} source needs the path its code lives at, and none was given. If this "
|
|
247
|
+
"agent has no code on disk, it is not this kind of source."
|
|
248
|
+
)
|
|
249
|
+
return _REGISTRY[kind](**kwargs)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def supported() -> tuple[str, ...]:
|
|
253
|
+
return tuple(sorted(_REGISTRY))
|
fi/alk/harness/spend.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""What the harness itself spent, written where a deleted sandbox cannot take it with it.
|
|
2
|
+
|
|
3
|
+
Every model call the harness makes arrives at one place, `Stage`'s handling of `StageDone`, so the
|
|
4
|
+
ledger is fed from there rather than from each stage's own code: a writer added later is counted
|
|
5
|
+
without anybody remembering to count it. Parallel scenario writers and the suite review each open
|
|
6
|
+
their own session, which is why per-call-site accounting would have missed most of a large run's
|
|
7
|
+
spend.
|
|
8
|
+
|
|
9
|
+
The file is rewritten after every turn so the newest total is always on disk. A run that dies
|
|
10
|
+
mid-turn loses that turn only, and the platform reads the file while the sandbox is alive.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import tempfile
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
JOURNAL_ALIAS = "ALK_SPEND_JOURNAL"
|
|
22
|
+
|
|
23
|
+
_stages: dict[str, dict[str, Any]] = {}
|
|
24
|
+
_path: Path | None = None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def journal_to(path: str | os.PathLike[str] | None) -> None:
|
|
28
|
+
"""Where to keep the ledger. Nothing is written until this is set or the alias names a path."""
|
|
29
|
+
global _path
|
|
30
|
+
_path = Path(path) if path else None
|
|
31
|
+
if _path is not None:
|
|
32
|
+
_flush()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _destination() -> Path | None:
|
|
36
|
+
if _path is not None:
|
|
37
|
+
return _path
|
|
38
|
+
named = os.environ.get(JOURNAL_ALIAS, "").strip()
|
|
39
|
+
return Path(named) if named else None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def record(
|
|
43
|
+
stage: str,
|
|
44
|
+
usd: float | None,
|
|
45
|
+
turns: int = 0,
|
|
46
|
+
models: set[str] | None = None,
|
|
47
|
+
tokens_in: int = 0,
|
|
48
|
+
tokens_out: int = 0,
|
|
49
|
+
tokens_cached: int = 0,
|
|
50
|
+
) -> None:
|
|
51
|
+
"""Add one session's reported spend. A backend that cannot price a call reports None.
|
|
52
|
+
|
|
53
|
+
``tokens_cached`` is the part of ``tokens_in`` the provider served from its own cache. It is
|
|
54
|
+
reported rather than discounted, because the table here carries no cache rate and a guessed
|
|
55
|
+
one would be a made-up figure presented as a price. Carrying the count is what lets anyone
|
|
56
|
+
reconciling a bill see the size of the overstatement instead of inheriting it silently: two
|
|
57
|
+
reruns of the same authoring produced byte-identical ledgers four times apart in wall clock,
|
|
58
|
+
which is what caching looks like when nothing records it.
|
|
59
|
+
"""
|
|
60
|
+
name = (stage or "stage").strip() or "stage"
|
|
61
|
+
entry = _stages.setdefault(
|
|
62
|
+
name,
|
|
63
|
+
{
|
|
64
|
+
"usd": 0.0,
|
|
65
|
+
"turns": 0,
|
|
66
|
+
"models": [],
|
|
67
|
+
"priced": 0,
|
|
68
|
+
"unpriced": 0,
|
|
69
|
+
"tokens_in": 0,
|
|
70
|
+
"tokens_out": 0,
|
|
71
|
+
"tokens_cached": 0,
|
|
72
|
+
},
|
|
73
|
+
)
|
|
74
|
+
if usd is None:
|
|
75
|
+
entry["unpriced"] += 1
|
|
76
|
+
else:
|
|
77
|
+
entry["usd"] = round(entry["usd"] + float(usd), 6)
|
|
78
|
+
entry["priced"] += 1
|
|
79
|
+
entry["turns"] += int(turns or 0)
|
|
80
|
+
entry["tokens_in"] += int(tokens_in or 0)
|
|
81
|
+
entry["tokens_out"] += int(tokens_out or 0)
|
|
82
|
+
entry["tokens_cached"] += int(tokens_cached or 0)
|
|
83
|
+
for model in sorted(models or set()):
|
|
84
|
+
if model not in entry["models"]:
|
|
85
|
+
entry["models"].append(model)
|
|
86
|
+
_flush()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def total_usd() -> float:
|
|
90
|
+
return round(sum(float(entry["usd"]) for entry in _stages.values()), 6)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def unpriced_turns() -> int:
|
|
94
|
+
"""Turns whose backend reported no price. Nonzero means the total is a floor, not the answer."""
|
|
95
|
+
return sum(int(entry["unpriced"]) for entry in _stages.values())
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def snapshot() -> dict[str, Any]:
|
|
99
|
+
return {
|
|
100
|
+
"schema": "futureagi.harness-spend.v1",
|
|
101
|
+
"total_usd": total_usd(),
|
|
102
|
+
"unpriced_turns": unpriced_turns(),
|
|
103
|
+
"stages": [
|
|
104
|
+
{
|
|
105
|
+
"stage": name,
|
|
106
|
+
**{
|
|
107
|
+
key: entry[key]
|
|
108
|
+
for key in (
|
|
109
|
+
"usd",
|
|
110
|
+
"turns",
|
|
111
|
+
"models",
|
|
112
|
+
"priced",
|
|
113
|
+
"unpriced",
|
|
114
|
+
"tokens_in",
|
|
115
|
+
"tokens_out",
|
|
116
|
+
"tokens_cached",
|
|
117
|
+
)
|
|
118
|
+
},
|
|
119
|
+
}
|
|
120
|
+
for name, entry in sorted(_stages.items())
|
|
121
|
+
],
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _flush() -> None:
|
|
126
|
+
destination = _destination()
|
|
127
|
+
if destination is None:
|
|
128
|
+
return
|
|
129
|
+
try:
|
|
130
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
131
|
+
# Written whole, then moved: a poll that reads mid-write must never see half a total.
|
|
132
|
+
handle = tempfile.NamedTemporaryFile(
|
|
133
|
+
"w", dir=destination.parent, prefix=".spend-", suffix=".json", delete=False
|
|
134
|
+
)
|
|
135
|
+
with handle as writing:
|
|
136
|
+
json.dump(snapshot(), writing, indent=2, sort_keys=True)
|
|
137
|
+
os.replace(handle.name, destination)
|
|
138
|
+
except OSError:
|
|
139
|
+
# Accounting must never be the reason a stage fails.
|
|
140
|
+
pass
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Bundle-owned observable HTTP tool proxy used by hosted process environments."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import time
|
|
8
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
9
|
+
from urllib import error, request
|
|
10
|
+
|
|
11
|
+
import psycopg
|
|
12
|
+
|
|
13
|
+
PORT = int(os.environ["PORT"])
|
|
14
|
+
UPSTREAM = os.environ["UPSTREAM_URL"].rstrip("/")
|
|
15
|
+
DATABASE_URL = os.environ["DATABASE_URL"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _record(
|
|
19
|
+
name: str, arguments: object, result: object, ok: bool, failure: str = ""
|
|
20
|
+
) -> None:
|
|
21
|
+
try:
|
|
22
|
+
with psycopg.connect(DATABASE_URL, autocommit=True) as connection:
|
|
23
|
+
connection.execute(
|
|
24
|
+
"INSERT INTO _alk_tool_trace(name, arguments, result, ok, error, at) "
|
|
25
|
+
"VALUES (%s, %s, %s, %s, %s, %s)",
|
|
26
|
+
(
|
|
27
|
+
name,
|
|
28
|
+
json.dumps(arguments),
|
|
29
|
+
json.dumps(result),
|
|
30
|
+
ok,
|
|
31
|
+
failure,
|
|
32
|
+
time.time(),
|
|
33
|
+
),
|
|
34
|
+
)
|
|
35
|
+
except Exception:
|
|
36
|
+
# Evidence persistence must never alter the target tool response.
|
|
37
|
+
pass
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Handler(BaseHTTPRequestHandler):
|
|
41
|
+
def log_message(self, *_args: object) -> None:
|
|
42
|
+
return
|
|
43
|
+
|
|
44
|
+
def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
45
|
+
if self.path == "/health":
|
|
46
|
+
self.send_response(200)
|
|
47
|
+
self.end_headers()
|
|
48
|
+
return
|
|
49
|
+
self._forward()
|
|
50
|
+
|
|
51
|
+
def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
52
|
+
self._forward()
|
|
53
|
+
|
|
54
|
+
def _forward(self) -> None:
|
|
55
|
+
size = int(self.headers.get("content-length") or 0)
|
|
56
|
+
body = self.rfile.read(size) if size else b""
|
|
57
|
+
try:
|
|
58
|
+
arguments = json.loads(body) if body else {}
|
|
59
|
+
except ValueError:
|
|
60
|
+
arguments = {"_raw": body.decode("utf-8", errors="replace")}
|
|
61
|
+
outgoing = request.Request(
|
|
62
|
+
UPSTREAM + self.path,
|
|
63
|
+
data=body if self.command != "GET" else None,
|
|
64
|
+
method=self.command,
|
|
65
|
+
headers={
|
|
66
|
+
"content-type": self.headers.get("content-type", "application/json")
|
|
67
|
+
},
|
|
68
|
+
)
|
|
69
|
+
name = self.path.split("?", 1)[0].rstrip("/").rsplit("/", 1)[-1] or "unknown"
|
|
70
|
+
try:
|
|
71
|
+
with request.urlopen(outgoing, timeout=30) as response:
|
|
72
|
+
content = response.read()
|
|
73
|
+
status = int(response.status)
|
|
74
|
+
response_type = response.headers.get("content-type", "application/json")
|
|
75
|
+
try:
|
|
76
|
+
result = json.loads(content) if content else None
|
|
77
|
+
except ValueError:
|
|
78
|
+
result = content.decode("utf-8", errors="replace")
|
|
79
|
+
_record(name, arguments, result, status < 400)
|
|
80
|
+
self.send_response(status)
|
|
81
|
+
self.send_header("content-type", response_type)
|
|
82
|
+
self.send_header("content-length", str(len(content)))
|
|
83
|
+
self.end_headers()
|
|
84
|
+
self.wfile.write(content)
|
|
85
|
+
except error.HTTPError as exc:
|
|
86
|
+
content = exc.read()
|
|
87
|
+
failure = content.decode("utf-8", errors="replace")[:2000]
|
|
88
|
+
_record(name, arguments, None, False, failure)
|
|
89
|
+
self.send_response(exc.code)
|
|
90
|
+
self.send_header("content-length", str(len(content)))
|
|
91
|
+
self.end_headers()
|
|
92
|
+
self.wfile.write(content)
|
|
93
|
+
except Exception as exc:
|
|
94
|
+
_record(name, arguments, None, False, f"{type(exc).__name__}: unavailable")
|
|
95
|
+
content = json.dumps({"detail": "tool_upstream_unavailable"}).encode()
|
|
96
|
+
self.send_response(502)
|
|
97
|
+
self.send_header("content-type", "application/json")
|
|
98
|
+
self.send_header("content-length", str(len(content)))
|
|
99
|
+
self.end_headers()
|
|
100
|
+
self.wfile.write(content)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
if __name__ == "__main__":
|
|
104
|
+
ThreadingHTTPServer(("127.0.0.1", PORT), Handler).serve_forever()
|