agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Account-integrated telemetry (Phase 8): local run ledger + keyed sync.
|
|
2
|
+
|
|
3
|
+
Two channels, never a third (P8-D1): (a) the always-on local run ledger —
|
|
4
|
+
every kit run appends one content-addressed, hash-chained row to
|
|
5
|
+
``${AGENT_LEARNING_HOME:-~/.agent-learning}/ledger/runs.jsonl``; (b) keyed
|
|
6
|
+
sync to the USER'S OWN Future AGI account when their keys resolve. There is
|
|
7
|
+
no anonymous analytics channel anywhere in the kit — structurally absent and
|
|
8
|
+
gate-proven (``telemetry_boundary``, gate #72).
|
|
9
|
+
|
|
10
|
+
Module scope imports are stdlib + the stdlib-only package internals; the
|
|
11
|
+
network-capable sync lane (``_sync``) is imported lazily inside functions
|
|
12
|
+
only, after the kill switch and key gates. ``AGENT_LEARNING_TELEMETRY=off``
|
|
13
|
+
binds everything, including vendored ``fi/*`` (P8-D6).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import Any, Mapping
|
|
19
|
+
|
|
20
|
+
from ._contract import ( # noqa: F401 — package canon re-exports
|
|
21
|
+
AGENT_LEARNING_RUN_KIND,
|
|
22
|
+
EVIDENCE_CLASSES,
|
|
23
|
+
GAP_SCHEMA,
|
|
24
|
+
GENESIS,
|
|
25
|
+
LEDGER_DIR_NAME,
|
|
26
|
+
LEDGER_HOME_ENV,
|
|
27
|
+
LEDGER_PATH_ENV,
|
|
28
|
+
LEDGER_ROW_SCHEMA,
|
|
29
|
+
NON_CANONICAL_FIELDS,
|
|
30
|
+
PHASES,
|
|
31
|
+
RELEASE_ADMISSIBLE_EVIDENCE_CLASSES,
|
|
32
|
+
ROW_FIELDS,
|
|
33
|
+
SYNC_MODE_AUTO,
|
|
34
|
+
SYNC_MODE_ENV,
|
|
35
|
+
SYNC_MODE_LOCAL,
|
|
36
|
+
SYNC_MODES,
|
|
37
|
+
SYNC_STATES,
|
|
38
|
+
TELEMETRY_ENV,
|
|
39
|
+
TELEMETRY_OFF_VALUE,
|
|
40
|
+
TOMBSTONE_FIELDS,
|
|
41
|
+
TOMBSTONE_REASONS,
|
|
42
|
+
TOMBSTONE_SCHEMA,
|
|
43
|
+
UNREADABLE_LINE_SCHEMA,
|
|
44
|
+
VERDICTS,
|
|
45
|
+
kill_switch_on,
|
|
46
|
+
ledger_dir,
|
|
47
|
+
sync_mode,
|
|
48
|
+
)
|
|
49
|
+
from ._ledger import RunLedger # noqa: F401
|
|
50
|
+
from ._queue import TelemetryQueue, global_queue # noqa: F401
|
|
51
|
+
from ._row import ( # noqa: F401
|
|
52
|
+
build_ledger_row,
|
|
53
|
+
canonical_row_address,
|
|
54
|
+
canonical_row_bytes,
|
|
55
|
+
content_admissible,
|
|
56
|
+
declared_required_env,
|
|
57
|
+
)
|
|
58
|
+
from ._run import ( # noqa: F401
|
|
59
|
+
RunRecorder,
|
|
60
|
+
RunSummary,
|
|
61
|
+
emit_run,
|
|
62
|
+
run_telemetry,
|
|
63
|
+
)
|
|
64
|
+
from ._url import build_dashboard_url # noqa: F401
|
|
65
|
+
|
|
66
|
+
__all__ = [
|
|
67
|
+
"AGENT_LEARNING_RUN_KIND",
|
|
68
|
+
"EVIDENCE_CLASSES",
|
|
69
|
+
"GAP_SCHEMA",
|
|
70
|
+
"GENESIS",
|
|
71
|
+
"LEDGER_DIR_NAME",
|
|
72
|
+
"LEDGER_HOME_ENV",
|
|
73
|
+
"LEDGER_PATH_ENV",
|
|
74
|
+
"LEDGER_ROW_SCHEMA",
|
|
75
|
+
"NON_CANONICAL_FIELDS",
|
|
76
|
+
"PHASES",
|
|
77
|
+
"RELEASE_ADMISSIBLE_EVIDENCE_CLASSES",
|
|
78
|
+
"ROW_FIELDS",
|
|
79
|
+
"RunLedger",
|
|
80
|
+
"RunRecorder",
|
|
81
|
+
"RunSummary",
|
|
82
|
+
"SYNC_MODE_AUTO",
|
|
83
|
+
"SYNC_MODE_ENV",
|
|
84
|
+
"SYNC_MODE_LOCAL",
|
|
85
|
+
"SYNC_MODES",
|
|
86
|
+
"SYNC_STATES",
|
|
87
|
+
"TELEMETRY_ENV",
|
|
88
|
+
"TELEMETRY_OFF_VALUE",
|
|
89
|
+
"TOMBSTONE_FIELDS",
|
|
90
|
+
"TOMBSTONE_REASONS",
|
|
91
|
+
"TOMBSTONE_SCHEMA",
|
|
92
|
+
"TelemetryQueue",
|
|
93
|
+
"UNREADABLE_LINE_SCHEMA",
|
|
94
|
+
"VERDICTS",
|
|
95
|
+
"build_dashboard_url",
|
|
96
|
+
"build_ledger_row",
|
|
97
|
+
"canonical_row_address",
|
|
98
|
+
"canonical_row_bytes",
|
|
99
|
+
"content_admissible",
|
|
100
|
+
"declared_required_env",
|
|
101
|
+
"emit_run",
|
|
102
|
+
"flush",
|
|
103
|
+
"kill_switch_on",
|
|
104
|
+
"ledger_dir",
|
|
105
|
+
"record_run",
|
|
106
|
+
"run_telemetry",
|
|
107
|
+
"sync_mode",
|
|
108
|
+
]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _handle_row(row: Mapping[str, Any], dropped: int) -> None:
|
|
112
|
+
"""Drain-side handler: ledger append (+ gap marker for any drops since
|
|
113
|
+
the last successful append). Runs on the worker thread; every failure is
|
|
114
|
+
swallowed by the queue (R§3.5).
|
|
115
|
+
|
|
116
|
+
Sync is EXPLICIT in v1 — ``agent-learn runs sync [<id>|--queued]`` or the
|
|
117
|
+
SDK ``telemetry._sync.sync_run`` — never fired from the emission path.
|
|
118
|
+
Emission-time auto-sync would turn every stray key in the environment
|
|
119
|
+
(test/example dummies included) into a network attempt inside release
|
|
120
|
+
flows; rows queue locally instead and the queued-sync path is idempotent
|
|
121
|
+
by content address, so nothing is ever lost (P8-D3, R§3.5).
|
|
122
|
+
"""
|
|
123
|
+
|
|
124
|
+
ledger = RunLedger()
|
|
125
|
+
if dropped > 0:
|
|
126
|
+
ledger.append_gap(dropped)
|
|
127
|
+
ledger.append(row)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def record_run(run_payload: Mapping[str, Any]) -> None:
|
|
131
|
+
"""The single emission hook target (ARCH Decision 7): called once at the
|
|
132
|
+
run-manifest boundary for every ``agent-learning.run.v1`` payload.
|
|
133
|
+
|
|
134
|
+
Out of the critical path: builds the redacted, content-addressed row and
|
|
135
|
+
does an O(1) bounded enqueue. ``AGENT_LEARNING_TELEMETRY=off`` disables
|
|
136
|
+
everything — ledger append and sync alike (P8-D6).
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
if kill_switch_on():
|
|
140
|
+
return
|
|
141
|
+
required_env = declared_required_env(run_payload)
|
|
142
|
+
row = build_ledger_row(run_payload, required_env=required_env)
|
|
143
|
+
global_queue(_handle_row).enqueue(row)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def flush(timeout: float = 5.0) -> bool:
|
|
147
|
+
"""Best-effort drain of the emission queue (atexit calls this too)."""
|
|
148
|
+
|
|
149
|
+
return global_queue(_handle_row).flush(timeout)
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Telemetry-local contract: row schemas, ledger paths, kill switch (Phase 8).
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only, plus the two reused live/ seams. Evidence classes and
|
|
4
|
+
the run kind are IMPORTED from ``live/_contract.py`` and never redeclared —
|
|
5
|
+
one vocabulary, one redaction seam (ARCH §2.0; the VS-Code "components exempt
|
|
6
|
+
from the off switch" failure mode is structurally impossible with one seam).
|
|
7
|
+
The ``telemetry_boundary`` gate scans this package like any release module.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
# --- reused verbatim from live/ (the ONE vocabulary; never redeclared) ------
|
|
16
|
+
from ..live._contract import ( # noqa: F401 — re-exported package canon
|
|
17
|
+
AGENT_LEARNING_RUN_KIND,
|
|
18
|
+
EVIDENCE_CLASSES,
|
|
19
|
+
RELEASE_ADMISSIBLE_EVIDENCE_CLASSES,
|
|
20
|
+
VERDICTS,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# --- row schema tags (ARCH §3; REVIEW-RULINGS MF3) ---------------------------
|
|
24
|
+
LEDGER_ROW_SCHEMA = "agent-learning.ledger-row.v1"
|
|
25
|
+
TOMBSTONE_SCHEMA = "agent-learning.ledger-tombstone.v1"
|
|
26
|
+
GAP_SCHEMA = "agent-learning.ledger-gap.v1"
|
|
27
|
+
UNREADABLE_LINE_SCHEMA = "agent-learning.ledger-unreadable-line.v1"
|
|
28
|
+
|
|
29
|
+
# --- hash-chain genesis sentinel (ARCH §2b; MF4: self-describing string) ----
|
|
30
|
+
GENESIS = "agent-learning.ledger.genesis.v1"
|
|
31
|
+
|
|
32
|
+
# --- semconv pin (PRD §4.4: OTel GenAI semconv is Development mid-2026) -----
|
|
33
|
+
SEMCONV_VERSION_ENV = "OTEL_SEMCONV_STABILITY_OPT_IN"
|
|
34
|
+
|
|
35
|
+
# --- kill switch (P8-D6: binds EVERYTHING including vendored fi/*) ----------
|
|
36
|
+
TELEMETRY_ENV = "AGENT_LEARNING_TELEMETRY"
|
|
37
|
+
TELEMETRY_OFF_VALUE = "off"
|
|
38
|
+
|
|
39
|
+
# --- sync mode (Phase 14, W&B `WANDB_MODE` analogue) -------------------------
|
|
40
|
+
# Reconciles the user's "keys present -> dashboard" (W&B-online) intent with the
|
|
41
|
+
# P8 doctrine that emission must NOT auto-sync in release/CI flows. ``auto``
|
|
42
|
+
# (default) emits when keys resolve; ``local`` queues locally + explicit
|
|
43
|
+
# ``runs sync`` only. The kill switch overrides both. The test/gate harness
|
|
44
|
+
# pins ``local`` so no internal flow makes a surprise network call.
|
|
45
|
+
SYNC_MODE_ENV = "AGENT_LEARNING_SYNC"
|
|
46
|
+
SYNC_MODE_AUTO = "auto"
|
|
47
|
+
SYNC_MODE_LOCAL = "local"
|
|
48
|
+
SYNC_MODES = (SYNC_MODE_AUTO, SYNC_MODE_LOCAL)
|
|
49
|
+
|
|
50
|
+
# --- ledger disk layout (ARCH §2a / Decision 9; MF5) -------------------------
|
|
51
|
+
LEDGER_HOME_ENV = "AGENT_LEARNING_HOME"
|
|
52
|
+
LEDGER_PATH_ENV = "AGENT_LEARNING_LEDGER_PATH" # overrides the DIRECTORY
|
|
53
|
+
LEDGER_DIR_NAME = "ledger"
|
|
54
|
+
ROWS_FILENAME = "runs.jsonl"
|
|
55
|
+
CHAIN_HEAD_FILENAME = "chain.head"
|
|
56
|
+
SYNC_CURSOR_FILENAME = "sync.cursor"
|
|
57
|
+
|
|
58
|
+
# --- canonical row field set (ARCH §2a table; MF3) ---------------------------
|
|
59
|
+
ROW_FIELDS = (
|
|
60
|
+
"schema",
|
|
61
|
+
"kind",
|
|
62
|
+
"phase",
|
|
63
|
+
"evidence_class",
|
|
64
|
+
"verdict",
|
|
65
|
+
"scores",
|
|
66
|
+
"gate_outcomes",
|
|
67
|
+
"semconv_version",
|
|
68
|
+
"manifest_address",
|
|
69
|
+
"asset_refs",
|
|
70
|
+
"trace_ids",
|
|
71
|
+
"content_bearing",
|
|
72
|
+
"redaction",
|
|
73
|
+
"created_at",
|
|
74
|
+
"run_id",
|
|
75
|
+
"chain",
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# fields excluded from the canonical hash preimage — they ARE the hash /
|
|
79
|
+
# envelope (ARCH §2a: the addressed core excludes exactly these three):
|
|
80
|
+
NON_CANONICAL_FIELDS = ("created_at", "run_id", "chain")
|
|
81
|
+
|
|
82
|
+
# --- tombstone fields (ARCH §2b; never rewrite a row — append) ---------------
|
|
83
|
+
TOMBSTONE_FIELDS = (
|
|
84
|
+
"schema",
|
|
85
|
+
"kind",
|
|
86
|
+
"tombstones",
|
|
87
|
+
"reason",
|
|
88
|
+
"redacted_fields",
|
|
89
|
+
"evidence_class",
|
|
90
|
+
"created_at",
|
|
91
|
+
"run_id",
|
|
92
|
+
"chain",
|
|
93
|
+
)
|
|
94
|
+
TOMBSTONE_REASONS = ("forget", "rollback", "redaction")
|
|
95
|
+
|
|
96
|
+
# --- workflow phase enum (ARCH §3) -------------------------------------------
|
|
97
|
+
PHASES = ("simulate", "evals", "optimize", "redteam", "suite", "live")
|
|
98
|
+
|
|
99
|
+
# --- sync attribute namespace (frozen with platform; PLATFORM-GROUNDING §6) --
|
|
100
|
+
FI_KIT_RUN_ID_ATTR = "fi.kit.run_id"
|
|
101
|
+
FI_KIT_PHASE_ATTR = "fi.kit.phase"
|
|
102
|
+
FI_KIT_WORLD_ATTR = "fi.kit.world"
|
|
103
|
+
|
|
104
|
+
# --- sync states the viewer renders (UI-UX §1.1 SYNCED column) ---------------
|
|
105
|
+
SYNC_STATES = ("local", "metadata", "metadata+content", "queued", "off")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def sync_mode() -> str:
|
|
109
|
+
"""The W&B-style telemetry mode (Phase 14). ``off`` when the kill switch is
|
|
110
|
+
set (it binds everything, P8-D6); otherwise ``AGENT_LEARNING_SYNC`` —
|
|
111
|
+
``auto`` (default: emit when keys resolve) or ``local`` (queue only)."""
|
|
112
|
+
|
|
113
|
+
if kill_switch_on():
|
|
114
|
+
return TELEMETRY_OFF_VALUE
|
|
115
|
+
value = os.environ.get(SYNC_MODE_ENV, "").strip().lower()
|
|
116
|
+
return value if value in SYNC_MODES else SYNC_MODE_AUTO
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def kill_switch_on() -> bool:
|
|
120
|
+
"""``AGENT_LEARNING_TELEMETRY=off`` disables ledger + sync, binding every
|
|
121
|
+
component including vendored ``fi/*`` (P8-D6; the gate's check 1 statically
|
|
122
|
+
verifies every emission path routes through this one guard)."""
|
|
123
|
+
|
|
124
|
+
return os.environ.get(TELEMETRY_ENV, "").strip().lower() == TELEMETRY_OFF_VALUE
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def ledger_dir(root: str | Path | None = None) -> Path:
|
|
128
|
+
"""Resolve the user-owned ledger DIRECTORY (ARCH Decision 9).
|
|
129
|
+
|
|
130
|
+
Precedence: explicit ``root`` arg > ``AGENT_LEARNING_LEDGER_PATH`` (the
|
|
131
|
+
directory override for tests/CI/per-project layouts) >
|
|
132
|
+
``${AGENT_LEARNING_HOME:-~/.agent-learning}/ledger/``.
|
|
133
|
+
"""
|
|
134
|
+
|
|
135
|
+
if root is not None:
|
|
136
|
+
return Path(root)
|
|
137
|
+
override = os.environ.get(LEDGER_PATH_ENV)
|
|
138
|
+
if override:
|
|
139
|
+
return Path(override)
|
|
140
|
+
home = os.environ.get(LEDGER_HOME_ENV) or (Path.home() / ".agent-learning")
|
|
141
|
+
return Path(home) / LEDGER_DIR_NAME
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""Export-result-aware OTLP emit (Phase 14, ARCH §2) — the truthful span sender.
|
|
2
|
+
|
|
3
|
+
THE FIX for the false-``synced`` bug (RESEARCH §1.2): the OTLP HTTP exporter
|
|
4
|
+
swallows export failures (it logs a 401 to stderr and returns
|
|
5
|
+
``SpanExportResult.FAILURE`` without raising), so the old ``register(...)`` path
|
|
6
|
+
completed its ``try`` block and reported ``synced`` while *nothing landed*. This
|
|
7
|
+
module owns the exporter via a recording wrapper, so a reported success is an
|
|
8
|
+
*observed* ``SpanExportResult.SUCCESS`` — never an assumption.
|
|
9
|
+
|
|
10
|
+
Vendored-engine boundary (gate #72): every ``fi_instrumentation`` import is
|
|
11
|
+
in-function. This module is imported only from inside the keyed branch of
|
|
12
|
+
``_run`` / ``_sync`` — never on the keyless import graph.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any, Mapping, Sequence
|
|
18
|
+
|
|
19
|
+
from ._contract import (
|
|
20
|
+
FI_KIT_PHASE_ATTR,
|
|
21
|
+
FI_KIT_RUN_ID_ATTR,
|
|
22
|
+
FI_KIT_WORLD_ATTR,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
# project_type/version/semconv resource attrs the collector resolves a project by
|
|
26
|
+
# (verified constants, fi_instrumentation/otel.py:57-89; RESEARCH §2.3).
|
|
27
|
+
_RES_PROJECT_NAME = "project_name"
|
|
28
|
+
_RES_PROJECT_TYPE = "project_type"
|
|
29
|
+
_RES_PROJECT_VERSION_NAME = "project_version_name"
|
|
30
|
+
_RES_PROJECT_VERSION_ID = "project_version_id"
|
|
31
|
+
_RES_EVAL_TAGS = "eval_tags"
|
|
32
|
+
_RES_METADATA = "metadata"
|
|
33
|
+
_RES_SEMCONV = "semantic_convention"
|
|
34
|
+
|
|
35
|
+
DEFAULT_PROJECT_NAME = "agent-learning"
|
|
36
|
+
RUN_SPAN_NAME = "agent-learning.run"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _recording_exporter(inner: Any) -> Any:
|
|
40
|
+
"""Wrap an OTLP SpanExporter so every ``export()`` result is recorded — the
|
|
41
|
+
single source of truth for 'did the span actually land'."""
|
|
42
|
+
|
|
43
|
+
from opentelemetry.sdk.trace.export import SpanExporter, SpanExportResult
|
|
44
|
+
|
|
45
|
+
class _RecordingExporter(SpanExporter): # type: ignore[misc]
|
|
46
|
+
def __init__(self, delegate: Any) -> None:
|
|
47
|
+
self._inner = delegate
|
|
48
|
+
self.results: list[Any] = []
|
|
49
|
+
|
|
50
|
+
def export(self, spans: Any) -> Any:
|
|
51
|
+
result = self._inner.export(spans)
|
|
52
|
+
self.results.append(result)
|
|
53
|
+
return result
|
|
54
|
+
|
|
55
|
+
def shutdown(self) -> Any:
|
|
56
|
+
return self._inner.shutdown()
|
|
57
|
+
|
|
58
|
+
def force_flush(self, timeout_millis: int = 30_000) -> bool:
|
|
59
|
+
flush = getattr(self._inner, "force_flush", None)
|
|
60
|
+
return flush(timeout_millis) if flush else True
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def ok(self) -> bool:
|
|
64
|
+
# success == at least one export, and EVERY export succeeded.
|
|
65
|
+
return bool(self.results) and all(
|
|
66
|
+
r is SpanExportResult.SUCCESS for r in self.results
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def last_reason(self) -> str:
|
|
71
|
+
if not self.results:
|
|
72
|
+
return "no_export_attempted"
|
|
73
|
+
return "ok" if self.ok else "export_rejected"
|
|
74
|
+
|
|
75
|
+
return _RecordingExporter(inner)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _build_provider(project_name: str, headers: Mapping[str, str]) -> tuple[Any, Any]:
|
|
79
|
+
"""Build a private tracer provider (NOT the global one — RESEARCH §4) with the
|
|
80
|
+
FI resource attrs the dashboard filters on, and swap in the recording exporter
|
|
81
|
+
so we can observe the real export result. Returns (provider, recorder)."""
|
|
82
|
+
|
|
83
|
+
import uuid
|
|
84
|
+
|
|
85
|
+
import fi_instrumentation.otel as fio
|
|
86
|
+
from fi_instrumentation.otel import SimpleSpanProcessor, Transport
|
|
87
|
+
from fi_instrumentation.settings import UuidIdGenerator
|
|
88
|
+
from opentelemetry.sdk.resources import Resource
|
|
89
|
+
|
|
90
|
+
# SimpleSpanProcessor(headers, transport) builds the correctly-configured OTLP
|
|
91
|
+
# exporter (endpoint resolved from FI_BASE_URL/BASE_URL). We then swap its
|
|
92
|
+
# exporter for the recording wrapper — reusing fi's construction, observing
|
|
93
|
+
# the result (the HTTPSpanExporter ctor itself does not accept transport).
|
|
94
|
+
processor = SimpleSpanProcessor(
|
|
95
|
+
headers=dict(headers), transport=Transport.HTTP
|
|
96
|
+
)
|
|
97
|
+
recorder = _recording_exporter(processor.span_exporter)
|
|
98
|
+
processor.span_exporter = recorder
|
|
99
|
+
|
|
100
|
+
resource = Resource(
|
|
101
|
+
attributes={
|
|
102
|
+
_RES_PROJECT_NAME: project_name,
|
|
103
|
+
_RES_PROJECT_TYPE: "experiment",
|
|
104
|
+
_RES_PROJECT_VERSION_NAME: DEFAULT_PROJECT_NAME,
|
|
105
|
+
_RES_PROJECT_VERSION_ID: str(uuid.uuid4()),
|
|
106
|
+
_RES_EVAL_TAGS: "[]",
|
|
107
|
+
_RES_METADATA: "{}",
|
|
108
|
+
_RES_SEMCONV: "fi",
|
|
109
|
+
}
|
|
110
|
+
)
|
|
111
|
+
provider = fio.TracerProvider(
|
|
112
|
+
resource=resource,
|
|
113
|
+
id_generator=UuidIdGenerator(),
|
|
114
|
+
transport=Transport.HTTP,
|
|
115
|
+
verbose=False,
|
|
116
|
+
)
|
|
117
|
+
provider.add_span_processor(processor)
|
|
118
|
+
return provider, recorder
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _scalar(value: Any) -> Any:
|
|
122
|
+
"""OTel span attributes accept str/bool/int/float (and homogeneous lists).
|
|
123
|
+
Coerce anything else to a compact string so a metric mapping never breaks
|
|
124
|
+
the emit."""
|
|
125
|
+
|
|
126
|
+
if isinstance(value, (str, bool, int, float)):
|
|
127
|
+
return value
|
|
128
|
+
import json
|
|
129
|
+
|
|
130
|
+
return json.dumps(value, sort_keys=True, separators=(",", ":"), default=str)[:1024]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def keyed_emit(
|
|
134
|
+
*,
|
|
135
|
+
span_name: str,
|
|
136
|
+
root_attrs: Mapping[str, Any],
|
|
137
|
+
children: Sequence[tuple[str, Mapping[str, Any]]] = (),
|
|
138
|
+
project_name: str,
|
|
139
|
+
headers: Mapping[str, str],
|
|
140
|
+
run_id: str = "",
|
|
141
|
+
phase: str = "",
|
|
142
|
+
world: str | None = None,
|
|
143
|
+
) -> dict[str, Any]:
|
|
144
|
+
"""Emit one run as a trace (root span + child spans) to the user's account and
|
|
145
|
+
report the OBSERVED export result.
|
|
146
|
+
|
|
147
|
+
Returns ``{status, trace_id, reason}`` where ``status`` is ``"synced"`` only
|
|
148
|
+
when the recording exporter saw ``SpanExportResult.SUCCESS``; otherwise
|
|
149
|
+
``"export_failed"`` with a reason. Any exception degrades to local
|
|
150
|
+
(``"deferred"``) — never propagates into the caller's run (R§3.5).
|
|
151
|
+
"""
|
|
152
|
+
|
|
153
|
+
try:
|
|
154
|
+
provider, recorder = _build_provider(project_name, headers)
|
|
155
|
+
tracer = provider.get_tracer("fi.alk.telemetry")
|
|
156
|
+
trace_id_hex = ""
|
|
157
|
+
with tracer.start_as_current_span(span_name) as root:
|
|
158
|
+
trace_id_hex = format(root.get_span_context().trace_id, "032x")
|
|
159
|
+
if run_id:
|
|
160
|
+
root.set_attribute(FI_KIT_RUN_ID_ATTR, run_id)
|
|
161
|
+
if phase:
|
|
162
|
+
root.set_attribute(FI_KIT_PHASE_ATTR, phase)
|
|
163
|
+
if world:
|
|
164
|
+
root.set_attribute(FI_KIT_WORLD_ATTR, str(world))
|
|
165
|
+
for key, value in root_attrs.items():
|
|
166
|
+
root.set_attribute(str(key), _scalar(value))
|
|
167
|
+
for child_name, child_attrs in children:
|
|
168
|
+
with tracer.start_as_current_span(str(child_name)) as child:
|
|
169
|
+
for key, value in (child_attrs or {}).items():
|
|
170
|
+
child.set_attribute(str(key), _scalar(value))
|
|
171
|
+
provider.force_flush()
|
|
172
|
+
provider.shutdown()
|
|
173
|
+
except BaseException as exc: # noqa: BLE001 — degrade-to-local (R§3.5)
|
|
174
|
+
return {"status": "deferred", "trace_id": None, "reason": f"{type(exc).__name__}: {exc}"}
|
|
175
|
+
|
|
176
|
+
if recorder.ok:
|
|
177
|
+
return {"status": "synced", "trace_id": trace_id_hex, "reason": None}
|
|
178
|
+
return {
|
|
179
|
+
"status": "export_failed",
|
|
180
|
+
"trace_id": None, # nothing landed → no viewable trace
|
|
181
|
+
"reason": recorder.last_reason,
|
|
182
|
+
}
|