agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/telemetry/_row.py
ADDED
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
"""Canonical ledger-row construction + serialization (Phase 8, ARCH ยง2a).
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only plus the two reused live/ seams. The content address must
|
|
4
|
+
be byte-identical on any machine, so serialization replicates the
|
|
5
|
+
``_schema.py:_json_sha256`` recipe exactly (``sort_keys=True``,
|
|
6
|
+
``separators=(",", ":")``, ``default=str``) โ one canonicalization discipline,
|
|
7
|
+
no second divergable serializer (ARCH Decision 2). Redaction runs BEFORE the
|
|
8
|
+
row is content-addressed or written: the address is computed over redacted
|
|
9
|
+
bytes, so a re-run that re-redacts produces the same address.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from typing import Any, Mapping, Sequence
|
|
18
|
+
|
|
19
|
+
from ..live._contract import AGENT_LEARNING_RUN_KIND, EVIDENCE_CLASSES, VERDICTS
|
|
20
|
+
from ..live._transcript import redact_env_values # the ONE redaction seam
|
|
21
|
+
from ._contract import (
|
|
22
|
+
LEDGER_ROW_SCHEMA,
|
|
23
|
+
NON_CANONICAL_FIELDS,
|
|
24
|
+
PHASES,
|
|
25
|
+
SEMCONV_VERSION_ENV,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Fixed precision for floats in the addressed core โ the kit's existing
|
|
29
|
+
# rounding rule (live/_transcript.py:105 rounds to 6 places).
|
|
30
|
+
_FLOAT_PRECISION = 6
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def canonical_row_bytes(row: Mapping[str, Any]) -> bytes:
|
|
34
|
+
"""The exact bytes ``run_id`` is the SHA-256 of (the addressed core).
|
|
35
|
+
|
|
36
|
+
Excludes ``created_at``/``run_id``/``chain`` โ and ONLY those three
|
|
37
|
+
(ARCH ยง2a): wall-clock and the chain digest are envelope fields that must
|
|
38
|
+
never enter the content address.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
preimage = {k: v for k, v in row.items() if k not in NON_CANONICAL_FIELDS}
|
|
42
|
+
return json.dumps(
|
|
43
|
+
preimage, sort_keys=True, separators=(",", ":"), default=str
|
|
44
|
+
).encode("utf-8") # == _schema.py:_json_sha256 recipe, byte-identical
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def canonical_row_address(row: Mapping[str, Any]) -> str:
|
|
48
|
+
"""``run_id = SHA-256(canonical addressed core)`` (P8-D3)."""
|
|
49
|
+
|
|
50
|
+
return hashlib.sha256(canonical_row_bytes(row)).hexdigest()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _redact_value(value: Any, required_env: Sequence[str]) -> Any:
|
|
54
|
+
"""Walk a row value: redact env VALUES out of every string leaf and round
|
|
55
|
+
floats to fixed precision so the addressed core has no platform-variant
|
|
56
|
+
repr (ARCH ยง2a determinism rules)."""
|
|
57
|
+
|
|
58
|
+
if isinstance(value, str):
|
|
59
|
+
return redact_env_values(value, required_env)
|
|
60
|
+
if isinstance(value, bool):
|
|
61
|
+
return value
|
|
62
|
+
if isinstance(value, float):
|
|
63
|
+
return round(value, _FLOAT_PRECISION)
|
|
64
|
+
if isinstance(value, Mapping):
|
|
65
|
+
return {
|
|
66
|
+
_redact_value(key, required_env): _redact_value(item, required_env)
|
|
67
|
+
for key, item in value.items()
|
|
68
|
+
}
|
|
69
|
+
if isinstance(value, (list, tuple)):
|
|
70
|
+
return [_redact_value(item, required_env) for item in value]
|
|
71
|
+
return value
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def build_ledger_row(
|
|
75
|
+
payload: Mapping[str, Any], *, required_env: Sequence[str] = ()
|
|
76
|
+
) -> dict[str, Any]:
|
|
77
|
+
"""Project an ``agent-learning.run.v1`` payload into a small ledger row of
|
|
78
|
+
metadata + content-addressed asset REFERENCES, never copies (PRD ยง4.1).
|
|
79
|
+
|
|
80
|
+
Redaction-before-serialize is the load-bearing ordering: ``_redact_value``
|
|
81
|
+
runs on the last step before ``canonical_row_address`` and before any disk
|
|
82
|
+
write โ the same seam+placement as ``live/_transcript.py:111``.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
summary = payload.get("summary")
|
|
86
|
+
summary = summary if isinstance(summary, Mapping) else {}
|
|
87
|
+
evidence_class = payload.get("evidence_class")
|
|
88
|
+
if evidence_class not in EVIDENCE_CLASSES:
|
|
89
|
+
evidence_class = "local_gate" # absence => local_gate (BUILD ยง1.3)
|
|
90
|
+
capture = payload.get("capture")
|
|
91
|
+
capture = capture if isinstance(capture, Mapping) else {}
|
|
92
|
+
row: dict[str, Any] = {
|
|
93
|
+
"schema": LEDGER_ROW_SCHEMA,
|
|
94
|
+
"kind": AGENT_LEARNING_RUN_KIND, # always the canonical run kind
|
|
95
|
+
"phase": _infer_phase(payload),
|
|
96
|
+
"evidence_class": evidence_class,
|
|
97
|
+
"verdict": _project_verdict(payload, summary),
|
|
98
|
+
"scores": _project_scores(summary),
|
|
99
|
+
"gate_outcomes": _project_gate_outcomes(payload),
|
|
100
|
+
"semconv_version": os.environ.get(SEMCONV_VERSION_ENV) or "unset",
|
|
101
|
+
"manifest_address": _manifest_address(payload),
|
|
102
|
+
# ASSET REFERENCES โ content addresses, never copies (Rยง3.3):
|
|
103
|
+
"asset_refs": _asset_refs(payload),
|
|
104
|
+
"trace_ids": _trace_ids(payload),
|
|
105
|
+
"content_bearing": _content_bearing(payload, capture),
|
|
106
|
+
"redaction": _redaction_contract(capture),
|
|
107
|
+
}
|
|
108
|
+
# Redact env VALUES out of every string field BEFORE the row is
|
|
109
|
+
# content-addressed or written (Rยง1 2507.06350; PRD ยง4.1):
|
|
110
|
+
row = _redact_value(row, tuple(required_env))
|
|
111
|
+
row["run_id"] = canonical_row_address(row) # address AFTER redaction
|
|
112
|
+
return row # created_at/chain are added by the ledger append (envelope)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def content_admissible(run_payload: Mapping[str, Any]) -> bool:
|
|
116
|
+
"""The content-sync admission predicate (PRD ยง4.2): the same
|
|
117
|
+
``capture.redaction`` non-empty mapping + ``capture.reviewed is True``
|
|
118
|
+
shape the ``live_lane_boundary`` gate demands on captured fixtures."""
|
|
119
|
+
|
|
120
|
+
capture = run_payload.get("capture")
|
|
121
|
+
capture = capture if isinstance(capture, Mapping) else {}
|
|
122
|
+
redaction = capture.get("redaction")
|
|
123
|
+
has_map = isinstance(redaction, Mapping) and bool(redaction)
|
|
124
|
+
return has_map and capture.get("reviewed") is True
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def declared_required_env(payload: Mapping[str, Any]) -> tuple[str, ...]:
|
|
128
|
+
"""Collect declared env names from the run payload (names only โ the
|
|
129
|
+
redaction seam replaces their VALUES with ``[redacted:NAME]``)."""
|
|
130
|
+
|
|
131
|
+
names: list[str] = []
|
|
132
|
+
for source in (
|
|
133
|
+
payload.get("required_env"),
|
|
134
|
+
_mapping(payload.get("live_lane")).get("required_env"),
|
|
135
|
+
_mapping(payload.get("lane")).get("required_env"),
|
|
136
|
+
_mapping(_mapping(payload.get("capture")).get("redaction")),
|
|
137
|
+
):
|
|
138
|
+
if isinstance(source, Mapping):
|
|
139
|
+
names.extend(str(name) for name in source)
|
|
140
|
+
elif isinstance(source, (list, tuple)):
|
|
141
|
+
names.extend(str(name) for name in source)
|
|
142
|
+
seen: dict[str, None] = {}
|
|
143
|
+
for name in names:
|
|
144
|
+
if name:
|
|
145
|
+
seen.setdefault(name, None)
|
|
146
|
+
return tuple(seen)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _mapping(value: Any) -> dict[str, Any]:
|
|
150
|
+
return dict(value) if isinstance(value, Mapping) else {}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _infer_phase(payload: Mapping[str, Any]) -> str:
|
|
154
|
+
explicit = payload.get("phase")
|
|
155
|
+
if isinstance(explicit, str) and explicit in PHASES:
|
|
156
|
+
return explicit
|
|
157
|
+
if isinstance(payload.get("live_lane"), Mapping) or isinstance(
|
|
158
|
+
payload.get("lane"), (str, Mapping)
|
|
159
|
+
):
|
|
160
|
+
return "live"
|
|
161
|
+
if payload.get("optimization") is not None:
|
|
162
|
+
return "optimize"
|
|
163
|
+
if payload.get("redteam") is not None or payload.get("attacks") is not None:
|
|
164
|
+
return "redteam"
|
|
165
|
+
if payload.get("suite") is not None or payload.get("result_kinds") is not None:
|
|
166
|
+
return "suite"
|
|
167
|
+
if payload.get("evaluations") is not None or payload.get("evals") is not None:
|
|
168
|
+
return "evals"
|
|
169
|
+
return "simulate"
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _project_verdict(
|
|
173
|
+
payload: Mapping[str, Any], summary: Mapping[str, Any]
|
|
174
|
+
) -> str | None:
|
|
175
|
+
"""Echo the run's own verdict โ never recompute or reinterpret it
|
|
176
|
+
(ARCH ยง1.4: the ledger records the verdict it is handed)."""
|
|
177
|
+
|
|
178
|
+
for candidate in (payload.get("verdict"), summary.get("verdict")):
|
|
179
|
+
if isinstance(candidate, str) and candidate in VERDICTS:
|
|
180
|
+
return candidate
|
|
181
|
+
status = payload.get("status")
|
|
182
|
+
if status == "passed":
|
|
183
|
+
return "pass"
|
|
184
|
+
if status == "failed":
|
|
185
|
+
return "fail"
|
|
186
|
+
return None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _project_scores(summary: Mapping[str, Any]) -> dict[str, float]:
|
|
190
|
+
scores: dict[str, float] = {}
|
|
191
|
+
for key, value in summary.items():
|
|
192
|
+
if isinstance(value, bool):
|
|
193
|
+
continue
|
|
194
|
+
if isinstance(value, (int, float)):
|
|
195
|
+
scores[str(key)] = round(float(value), _FLOAT_PRECISION)
|
|
196
|
+
return scores
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _project_gate_outcomes(payload: Mapping[str, Any]) -> dict[str, bool]:
|
|
200
|
+
outcomes: dict[str, bool] = {}
|
|
201
|
+
declared = payload.get("gate_outcomes")
|
|
202
|
+
if isinstance(declared, Mapping):
|
|
203
|
+
for key, value in declared.items():
|
|
204
|
+
outcomes[str(key)] = bool(value)
|
|
205
|
+
return outcomes
|
|
206
|
+
checks = payload.get("checks")
|
|
207
|
+
if isinstance(checks, (list, tuple)):
|
|
208
|
+
for check in checks:
|
|
209
|
+
if isinstance(check, Mapping) and check.get("id") is not None:
|
|
210
|
+
outcomes[str(check["id"])] = bool(
|
|
211
|
+
check.get("passed", check.get("status") == "passed")
|
|
212
|
+
)
|
|
213
|
+
return outcomes
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _manifest_address(payload: Mapping[str, Any]) -> str | None:
|
|
217
|
+
manifest = payload.get("manifest")
|
|
218
|
+
if isinstance(manifest, Mapping) and manifest:
|
|
219
|
+
data = json.dumps(
|
|
220
|
+
manifest, sort_keys=True, separators=(",", ":"), default=str
|
|
221
|
+
).encode("utf-8")
|
|
222
|
+
return hashlib.sha256(data).hexdigest()
|
|
223
|
+
address = payload.get("manifest_address")
|
|
224
|
+
return str(address) if isinstance(address, str) and address else None
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _asset_refs(payload: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
228
|
+
refs: list[dict[str, Any]] = []
|
|
229
|
+
declared = payload.get("asset_refs")
|
|
230
|
+
if isinstance(declared, (list, tuple)):
|
|
231
|
+
for item in declared:
|
|
232
|
+
if not isinstance(item, Mapping):
|
|
233
|
+
continue
|
|
234
|
+
address = item.get("content_address") or item.get("content_hash")
|
|
235
|
+
if not address:
|
|
236
|
+
continue
|
|
237
|
+
ref: dict[str, Any] = {
|
|
238
|
+
"kind": str(item.get("kind") or "asset"),
|
|
239
|
+
"content_address": str(address),
|
|
240
|
+
}
|
|
241
|
+
if item.get("account_object_id"):
|
|
242
|
+
ref["account_object_id"] = str(item["account_object_id"])
|
|
243
|
+
refs.append(ref)
|
|
244
|
+
for plural, singular in (("personas", "persona"), ("scenarios", "scenario")):
|
|
245
|
+
for item in payload.get(plural) or []:
|
|
246
|
+
if not isinstance(item, Mapping):
|
|
247
|
+
continue
|
|
248
|
+
address = (
|
|
249
|
+
item.get("content_address")
|
|
250
|
+
or item.get("content_hash")
|
|
251
|
+
or item.get("version")
|
|
252
|
+
)
|
|
253
|
+
if not address:
|
|
254
|
+
continue
|
|
255
|
+
ref = {"kind": singular, "content_address": str(address)}
|
|
256
|
+
if item.get("account_object_id"):
|
|
257
|
+
ref["account_object_id"] = str(item["account_object_id"])
|
|
258
|
+
refs.append(ref)
|
|
259
|
+
return refs
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _trace_ids(payload: Mapping[str, Any]) -> list[str]:
|
|
263
|
+
declared = payload.get("trace_ids")
|
|
264
|
+
if isinstance(declared, (list, tuple)):
|
|
265
|
+
return [str(item) for item in declared if item]
|
|
266
|
+
return []
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _content_bearing(
|
|
270
|
+
payload: Mapping[str, Any], capture: Mapping[str, Any]
|
|
271
|
+
) -> bool:
|
|
272
|
+
"""True iff the row references captured content โ transcripts/prompts/
|
|
273
|
+
tool I/O (ARCH ยง2a); the sync content gate keys off it."""
|
|
274
|
+
|
|
275
|
+
if capture:
|
|
276
|
+
return True
|
|
277
|
+
if payload.get("transcripts"):
|
|
278
|
+
return True
|
|
279
|
+
declared = payload.get("asset_refs")
|
|
280
|
+
if isinstance(declared, (list, tuple)):
|
|
281
|
+
for item in declared:
|
|
282
|
+
if isinstance(item, Mapping) and item.get("kind") == "transcript":
|
|
283
|
+
return True
|
|
284
|
+
return False
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _redaction_contract(capture: Mapping[str, Any]) -> dict[str, Any] | None:
|
|
288
|
+
"""The capture+redaction mapping for content-bearing rows: env NAMES +
|
|
289
|
+
strategy โ names always, values never. ``None`` on metadata-only rows."""
|
|
290
|
+
|
|
291
|
+
redaction = capture.get("redaction")
|
|
292
|
+
if isinstance(redaction, Mapping) and redaction:
|
|
293
|
+
return {str(name): str(strategy) for name, strategy in redaction.items()}
|
|
294
|
+
return None
|
fi/alk/telemetry/_run.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""``run_telemetry`` (Phase 14, ARCH ยง4) โ the ONE telemetry surface every kit run
|
|
2
|
+
wraps its body in. The W&B / promptfoo model:
|
|
3
|
+
|
|
4
|
+
* local path โ ALWAYS: append the ledger row + return/print a RunSummary. No
|
|
5
|
+
network. Works credential-free (promptfoo-local).
|
|
6
|
+
* cloud path โ ADDITIVE, only when keys resolve and the collector is reachable:
|
|
7
|
+
emit the run as a real trace and print a clickable dashboard URL (W&B
|
|
8
|
+
"View run at โฆ"). The URL is printed ONLY on an OBSERVED export success.
|
|
9
|
+
|
|
10
|
+
Logs go to STDERR (W&B convention) so a kit run's STDOUT stays clean (the gate
|
|
11
|
+
example asserts empty stdout). Keys gate the destination, never the capability
|
|
12
|
+
(P8-D2); ``AGENT_LEARNING_TELEMETRY=off`` binds everything (P8-D6).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import sys
|
|
18
|
+
from contextlib import contextmanager
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from typing import Any, Iterator, Mapping
|
|
21
|
+
|
|
22
|
+
from ..config import AgentLearningConfig
|
|
23
|
+
from ._contract import SYNC_MODE_AUTO, kill_switch_on, ledger_dir, sync_mode
|
|
24
|
+
from ._ledger import RunLedger
|
|
25
|
+
from ._row import build_ledger_row
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class RunSummary:
|
|
30
|
+
"""What a kit run reports โ the value the caller attaches to its result and
|
|
31
|
+
``_log_summary`` renders."""
|
|
32
|
+
|
|
33
|
+
kind: str
|
|
34
|
+
name: str
|
|
35
|
+
status: str = "local" # local | synced | export_failed | deferred | off
|
|
36
|
+
metrics: dict[str, Any] = field(default_factory=dict)
|
|
37
|
+
run_id: str | None = None
|
|
38
|
+
trace_id: str | None = None
|
|
39
|
+
dashboard_url: str | None = None
|
|
40
|
+
url_kind: str | None = None
|
|
41
|
+
reason: str | None = None
|
|
42
|
+
|
|
43
|
+
def as_dict(self) -> dict[str, Any]:
|
|
44
|
+
return {
|
|
45
|
+
"kind": self.kind,
|
|
46
|
+
"name": self.name,
|
|
47
|
+
"status": self.status,
|
|
48
|
+
"metrics": dict(self.metrics),
|
|
49
|
+
"run_id": self.run_id,
|
|
50
|
+
"trace_id": self.trace_id,
|
|
51
|
+
"dashboard_url": self.dashboard_url,
|
|
52
|
+
"url_kind": self.url_kind,
|
|
53
|
+
"reason": self.reason,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class RunRecorder:
|
|
58
|
+
"""Collects metrics + child-span specs during a run; ``summary`` is populated
|
|
59
|
+
on context exit and readable by the caller afterwards."""
|
|
60
|
+
|
|
61
|
+
def __init__(self, kind: str, name: str, world: str | None = None) -> None:
|
|
62
|
+
self.kind = kind
|
|
63
|
+
self.name = name
|
|
64
|
+
self.world = world
|
|
65
|
+
self.metrics: dict[str, Any] = {}
|
|
66
|
+
self.children: list[tuple[str, dict[str, Any]]] = []
|
|
67
|
+
self.verdict: str | None = None
|
|
68
|
+
self.summary: RunSummary | None = None
|
|
69
|
+
|
|
70
|
+
def set_metrics(self, **metrics: Any) -> None:
|
|
71
|
+
self.metrics.update(metrics)
|
|
72
|
+
|
|
73
|
+
def set_verdict(self, verdict: str) -> None:
|
|
74
|
+
self.verdict = verdict
|
|
75
|
+
|
|
76
|
+
def add_child(self, name: str, attrs: Mapping[str, Any]) -> None:
|
|
77
|
+
self.children.append((str(name), dict(attrs)))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _ledger_payload(rec: RunRecorder) -> dict[str, Any]:
|
|
81
|
+
return {
|
|
82
|
+
"kind": rec.kind,
|
|
83
|
+
"summary": {
|
|
84
|
+
"name": rec.name,
|
|
85
|
+
"verdict": rec.verdict,
|
|
86
|
+
"metrics": rec.metrics,
|
|
87
|
+
},
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _log_summary(summary: RunSummary) -> None:
|
|
92
|
+
"""Render the W&B / promptfoo line to stderr."""
|
|
93
|
+
|
|
94
|
+
metric_bits = " ยท ".join(
|
|
95
|
+
f"{k} {v}" for k, v in summary.metrics.items()
|
|
96
|
+
if isinstance(v, (int, float, str))
|
|
97
|
+
)
|
|
98
|
+
head = f"agent-learning: {summary.kind} '{summary.name}'"
|
|
99
|
+
if metric_bits:
|
|
100
|
+
head += f" ยท {metric_bits}"
|
|
101
|
+
print(head, file=sys.stderr)
|
|
102
|
+
|
|
103
|
+
if summary.status == "synced" and summary.dashboard_url:
|
|
104
|
+
if summary.url_kind == "deep_link":
|
|
105
|
+
print(f"agent-learning: ๐ view in dashboard โ {summary.dashboard_url}", file=sys.stderr)
|
|
106
|
+
elif summary.url_kind == "project":
|
|
107
|
+
print(f"agent-learning: ๐ view in dashboard (project) โ {summary.dashboard_url}", file=sys.stderr)
|
|
108
|
+
else: # list_fallback
|
|
109
|
+
print(
|
|
110
|
+
f"agent-learning: ๐ dashboard โ {summary.dashboard_url} "
|
|
111
|
+
f"(find project '{summary.name}' / '{summary.metrics.get('project_name', '')}')",
|
|
112
|
+
file=sys.stderr,
|
|
113
|
+
)
|
|
114
|
+
elif summary.status == "export_failed":
|
|
115
|
+
print(
|
|
116
|
+
f"agent-learning: โ dashboard export not accepted ({summary.reason}) โ logged locally only",
|
|
117
|
+
file=sys.stderr,
|
|
118
|
+
)
|
|
119
|
+
elif summary.status == "deferred":
|
|
120
|
+
print(
|
|
121
|
+
f"agent-learning: dashboard unreachable ({summary.reason}) โ logged locally only",
|
|
122
|
+
file=sys.stderr,
|
|
123
|
+
)
|
|
124
|
+
elif summary.status == "off":
|
|
125
|
+
print("agent-learning: telemetry off โ nothing logged or sent", file=sys.stderr)
|
|
126
|
+
else: # local
|
|
127
|
+
print(f"agent-learning: logged locally โ {ledger_dir()}", file=sys.stderr)
|
|
128
|
+
print(
|
|
129
|
+
"agent-learning: set FI_API_KEY + FI_SECRET_KEY to view runs in the dashboard",
|
|
130
|
+
file=sys.stderr,
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _finalize(rec: RunRecorder, *, project_name: str | None) -> RunSummary:
|
|
135
|
+
summary = RunSummary(kind=rec.kind, name=rec.name, metrics=dict(rec.metrics))
|
|
136
|
+
|
|
137
|
+
# Kill switch binds EVERYTHING incl. the ledger (P8-D6).
|
|
138
|
+
if kill_switch_on():
|
|
139
|
+
summary.status = "off"
|
|
140
|
+
return summary
|
|
141
|
+
|
|
142
|
+
# Local path โ always (FR2).
|
|
143
|
+
row = build_ledger_row(_ledger_payload(rec))
|
|
144
|
+
try:
|
|
145
|
+
RunLedger().append(row)
|
|
146
|
+
except Exception: # noqa: BLE001 โ a ledger write failure must not break the run
|
|
147
|
+
pass
|
|
148
|
+
summary.run_id = row.get("run_id")
|
|
149
|
+
|
|
150
|
+
config = AgentLearningConfig.from_env()
|
|
151
|
+
# Cloud path requires keys AND mode=auto (W&B-online). mode=local (the test/
|
|
152
|
+
# gate default) queues locally โ no surprise network in release/CI (P8).
|
|
153
|
+
if not (config.api_key and config.secret_key) or sync_mode() != SYNC_MODE_AUTO:
|
|
154
|
+
summary.status = "local"
|
|
155
|
+
return summary
|
|
156
|
+
|
|
157
|
+
# Cloud path โ additive (FR3). Import network-capable code only here.
|
|
158
|
+
from . import _emit, _url
|
|
159
|
+
|
|
160
|
+
proj = project_name or "agent-learning"
|
|
161
|
+
summary.metrics.setdefault("project_name", proj)
|
|
162
|
+
emit = _emit.keyed_emit(
|
|
163
|
+
span_name=_emit.RUN_SPAN_NAME,
|
|
164
|
+
root_attrs={"kind": rec.kind, "name": rec.name, **rec.metrics},
|
|
165
|
+
children=rec.children,
|
|
166
|
+
project_name=proj,
|
|
167
|
+
headers={"X-Api-Key": config.api_key, "X-Secret-Key": config.secret_key},
|
|
168
|
+
run_id=summary.run_id or "",
|
|
169
|
+
phase=rec.kind,
|
|
170
|
+
world=rec.world,
|
|
171
|
+
)
|
|
172
|
+
summary.status = emit["status"]
|
|
173
|
+
summary.reason = emit.get("reason")
|
|
174
|
+
if emit["status"] == "synced":
|
|
175
|
+
summary.trace_id = emit.get("trace_id")
|
|
176
|
+
url = _url.build_dashboard_url(proj, summary.trace_id, config=config)
|
|
177
|
+
summary.dashboard_url = url["url"]
|
|
178
|
+
summary.url_kind = url["kind"]
|
|
179
|
+
return summary
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def emit_run(
|
|
183
|
+
*,
|
|
184
|
+
kind: str,
|
|
185
|
+
name: str,
|
|
186
|
+
metrics: Mapping[str, Any] | None = None,
|
|
187
|
+
verdict: str | None = None,
|
|
188
|
+
children: list[tuple[str, dict[str, Any]]] | None = None,
|
|
189
|
+
world: str | None = None,
|
|
190
|
+
project_name: str | None = None,
|
|
191
|
+
) -> RunSummary:
|
|
192
|
+
"""Non-context entrypoint for code paths that have already computed their
|
|
193
|
+
results (``run_benchmark`` / ``optimize_against_dataset`` / ``improve_agent_
|
|
194
|
+
code``): finalize one run (local ledger always; cloud emit when mode=auto +
|
|
195
|
+
keys), log the W&B/promptfoo line, and return the summary. Never raises into
|
|
196
|
+
the caller."""
|
|
197
|
+
|
|
198
|
+
try:
|
|
199
|
+
rec = RunRecorder(kind=kind, name=name, world=world)
|
|
200
|
+
if metrics:
|
|
201
|
+
rec.set_metrics(**dict(metrics))
|
|
202
|
+
if verdict:
|
|
203
|
+
rec.set_verdict(verdict)
|
|
204
|
+
for child_name, attrs in children or []:
|
|
205
|
+
rec.add_child(child_name, attrs)
|
|
206
|
+
rec.summary = _finalize(rec, project_name=project_name)
|
|
207
|
+
_log_summary(rec.summary)
|
|
208
|
+
return rec.summary
|
|
209
|
+
except Exception: # noqa: BLE001 โ telemetry is a side-channel, never fatal
|
|
210
|
+
return RunSummary(kind=kind, name=name, status="local")
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
@contextmanager
|
|
214
|
+
def run_telemetry(
|
|
215
|
+
*,
|
|
216
|
+
kind: str,
|
|
217
|
+
name: str,
|
|
218
|
+
world: str | None = None,
|
|
219
|
+
project_name: str | None = None,
|
|
220
|
+
) -> Iterator[RunRecorder]:
|
|
221
|
+
"""Wrap a kit run. Yields a ``RunRecorder``; after the block, ``rec.summary``
|
|
222
|
+
holds the finalized ``RunSummary`` (status + dashboard URL). Telemetry never
|
|
223
|
+
raises into the wrapped run."""
|
|
224
|
+
|
|
225
|
+
rec = RunRecorder(kind=kind, name=name, world=world)
|
|
226
|
+
try:
|
|
227
|
+
yield rec
|
|
228
|
+
finally:
|
|
229
|
+
try:
|
|
230
|
+
rec.summary = _finalize(rec, project_name=project_name)
|
|
231
|
+
_log_summary(rec.summary)
|
|
232
|
+
except Exception: # noqa: BLE001 โ telemetry is a side-channel, never fatal
|
|
233
|
+
rec.summary = RunSummary(kind=kind, name=name, status="local")
|