agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Phase 13D — the Practice Loop trainer (facade only; mirrors live/ style).
|
|
2
|
+
|
|
3
|
+
Lazy exports so ``import fi.alk.practice`` stays cheap. The trainer
|
|
4
|
+
employs the existing 13C operators; it adds no new step API and emits standard
|
|
5
|
+
``agent-learning.run.v1`` rows through ``run_manifest``/``public_payload`` so
|
|
6
|
+
every episode lands a telemetry ledger row with zero new telemetry code.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import importlib
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
# public name → home submodule (resolved lazily)
|
|
14
|
+
_LAZY_EXPORTS = {
|
|
15
|
+
# contract constants
|
|
16
|
+
"PRACTICE_PHASES": "_contract",
|
|
17
|
+
"PRACTICE_ARTIFACT_KINDS": "_contract",
|
|
18
|
+
"SCAFFOLD_TYPES": "_contract",
|
|
19
|
+
"LADDER_STATES": "_contract",
|
|
20
|
+
"PRACTICE_REPLAY_INTERVALS": "_contract",
|
|
21
|
+
"ZPD_BAND": "_contract",
|
|
22
|
+
"REVIEW_RATIO": "_contract",
|
|
23
|
+
"BUDGET_PLAN": "_contract",
|
|
24
|
+
"PRACTICE_STORE_ACTIVE_CAP": "_contract",
|
|
25
|
+
"SCAFFOLD_FADE_DEFAULT": "_contract",
|
|
26
|
+
"AGENT_LEARNING_PRACTICE_LOOP_KIND": "_contract",
|
|
27
|
+
"AGENT_LEARNING_PRACTICE_RESULT_KIND": "_contract",
|
|
28
|
+
"practice_store_path": "_contract",
|
|
29
|
+
# budget
|
|
30
|
+
"BudgetMeter": "_budget",
|
|
31
|
+
"BudgetExhausted": "_budget",
|
|
32
|
+
# trainer surface
|
|
33
|
+
"run_practice_loop": "_trainer",
|
|
34
|
+
"practice_report": "_assess",
|
|
35
|
+
"ladder_state": "_store",
|
|
36
|
+
"run_due_reviews": "_schedule",
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def __getattr__(name: str) -> Any:
|
|
41
|
+
home = _LAZY_EXPORTS.get(name)
|
|
42
|
+
if home is None:
|
|
43
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
44
|
+
module = importlib.import_module(f"{__name__}.{home}")
|
|
45
|
+
value = getattr(module, name)
|
|
46
|
+
globals()[name] = value
|
|
47
|
+
return value
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def __dir__() -> list[str]:
|
|
51
|
+
return sorted(set(globals()) | set(_LAZY_EXPORTS))
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Unit 10 (BBG U10 / ARCH §2d phase 1) — ASSESS: battery over the obligation grid.
|
|
2
|
+
|
|
3
|
+
Runs the battery over ``scenarios × cast × perturbations`` at ScenarioBinding
|
|
4
|
+
weights by deriving run manifests per cell (each scored episode charges the
|
|
5
|
+
meter), collecting verdict rows via loss.verdict_row, composing loss.loss_report.
|
|
6
|
+
Emits ``agent-learning.practice-report.v1`` through public_payload.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any, Callable, Dict, List, Mapping, Optional
|
|
11
|
+
|
|
12
|
+
from .._schema import public_payload
|
|
13
|
+
from .. import loss as _loss
|
|
14
|
+
from ._budget import BudgetMeter
|
|
15
|
+
from ._contract import AGENT_LEARNING_PRACTICE_REPORT_KIND
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _grid_cells(simulation: Mapping[str, Any]) -> List[dict]:
|
|
19
|
+
"""Enumerate obligation cells from the P7 CoverageDeclaration vocabulary;
|
|
20
|
+
degenerate single cell when no coverage declared."""
|
|
21
|
+
cells: List[dict] = []
|
|
22
|
+
for binding in simulation.get("scenarios") or []:
|
|
23
|
+
scenario = binding.get("scenario") or {}
|
|
24
|
+
coverage = scenario.get("coverage") or {}
|
|
25
|
+
intents = coverage.get("intents") or [None]
|
|
26
|
+
perturbations = coverage.get("perturbations") or [None]
|
|
27
|
+
for member in binding.get("cast") or []:
|
|
28
|
+
for intent in intents:
|
|
29
|
+
for perturbation in perturbations:
|
|
30
|
+
cells.append({
|
|
31
|
+
"intent": intent,
|
|
32
|
+
"persona": member.get("persona"),
|
|
33
|
+
"perturbation": perturbation,
|
|
34
|
+
"obligation": None,
|
|
35
|
+
"weight": float(binding.get("weight", 1.0)),
|
|
36
|
+
})
|
|
37
|
+
if not cells:
|
|
38
|
+
cells.append({"intent": None, "persona": None, "perturbation": None,
|
|
39
|
+
"obligation": None, "weight": 1.0})
|
|
40
|
+
return cells
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def assess(
|
|
44
|
+
simulation: Mapping[str, Any],
|
|
45
|
+
objective: Mapping[str, Any],
|
|
46
|
+
*,
|
|
47
|
+
meter: BudgetMeter,
|
|
48
|
+
round_no: int,
|
|
49
|
+
seed: int,
|
|
50
|
+
cell_scorer: Callable[[Mapping[str, Any]], Mapping[str, Any]],
|
|
51
|
+
parent_report_hash: Optional[str] = None,
|
|
52
|
+
repeats: int = 1,
|
|
53
|
+
coverage_source: str = "declared",
|
|
54
|
+
) -> dict:
|
|
55
|
+
"""Run the battery. ``cell_scorer(cell) -> {scalar, verdict, evidence_class}``
|
|
56
|
+
is the per-cell episode evaluator (injected for determinism/testing; in
|
|
57
|
+
production it derives + runs a run manifest). Each scored episode charges the
|
|
58
|
+
meter."""
|
|
59
|
+
cells = _grid_cells(simulation)
|
|
60
|
+
verdicts: List[dict] = []
|
|
61
|
+
calibration_mass_by_cell: Dict[str, float] = {}
|
|
62
|
+
for cell in cells:
|
|
63
|
+
for _ in range(max(1, int(repeats))):
|
|
64
|
+
meter.charge("assess", 1)
|
|
65
|
+
scored = cell_scorer(cell)
|
|
66
|
+
row = _loss.verdict_row(
|
|
67
|
+
eval_ref=scored.get("eval", "agent_report"),
|
|
68
|
+
cell=cell,
|
|
69
|
+
scalar=float(scored.get("scalar", 0.0)),
|
|
70
|
+
verdict=str(scored.get("verdict", "pass")),
|
|
71
|
+
evidence_class=str(scored.get("evidence_class", "local_gate")),
|
|
72
|
+
fidelity_admissible=bool(scored.get("fidelity_admissible", True)),
|
|
73
|
+
provenance={"round": round_no, "seed": seed},
|
|
74
|
+
)
|
|
75
|
+
verdicts.append(row)
|
|
76
|
+
if row["verdict"] == "unstable":
|
|
77
|
+
key = _loss._cell_key(cell)
|
|
78
|
+
calibration_mass_by_cell[key] = round(
|
|
79
|
+
calibration_mass_by_cell.get(key, 0.0) + 1.0, 6
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
loss_report = _loss.loss_report(objective, verdicts, budget_consumed=meter.consumed)
|
|
83
|
+
report = {
|
|
84
|
+
"kind": AGENT_LEARNING_PRACTICE_REPORT_KIND,
|
|
85
|
+
"round": int(round_no),
|
|
86
|
+
"objective_version": objective.get("version"),
|
|
87
|
+
"loss_report": loss_report,
|
|
88
|
+
"grid": {
|
|
89
|
+
"cells_total": len(cells),
|
|
90
|
+
"cells_assessed": len(cells),
|
|
91
|
+
"coverage_source": coverage_source,
|
|
92
|
+
},
|
|
93
|
+
"calibration_mass_by_cell": calibration_mass_by_cell,
|
|
94
|
+
"budget_consumed": meter.consumed,
|
|
95
|
+
"seed": int(seed),
|
|
96
|
+
"parent": parent_report_hash,
|
|
97
|
+
}
|
|
98
|
+
return public_payload(report, kind=AGENT_LEARNING_PRACTICE_REPORT_KIND)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def practice_report(*args: Any, **kwargs: Any) -> dict:
|
|
102
|
+
"""Public alias for the ASSESS report builder (facade export)."""
|
|
103
|
+
return assess(*args, **kwargs)
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Unit 8 (BBG U8 / ARCH §2d, AD-I) — the single budget meter.
|
|
2
|
+
|
|
3
|
+
ONE unit = one scored episode evaluation. Every assess row, ZPD repeat,
|
|
4
|
+
scaffolded/unscaffolded drill evaluation, inner-operator evaluation, scheduled
|
|
5
|
+
review row, and promotion-sweep row charges THIS meter — there is no second
|
|
6
|
+
currency. Soft per-phase enforcement of budget_plan with carry-over.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Dict
|
|
11
|
+
|
|
12
|
+
from ._contract import BUDGET_PLAN, PRACTICE_PHASES
|
|
13
|
+
|
|
14
|
+
# Map the 4-fraction budget_plan onto phases. assess / drill / update / review;
|
|
15
|
+
# diagnose+consolidate+calibrate draw from their adjacent phase allocations.
|
|
16
|
+
_BUDGET_PLAN_PHASES = ("assess", "drill", "update", "review")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class BudgetExhausted(RuntimeError):
|
|
20
|
+
"""Raised when the meter has no remaining budget (trainer stop)."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class BudgetMeter:
|
|
24
|
+
"""The single eval-unit meter (AD-I)."""
|
|
25
|
+
|
|
26
|
+
def __init__(self, total: int, *, budget_plan: tuple[float, ...] = BUDGET_PLAN) -> None:
|
|
27
|
+
if not isinstance(total, int) or isinstance(total, bool) or total < 1:
|
|
28
|
+
raise ValueError("budget total must be an int >= 1")
|
|
29
|
+
self.total = int(total)
|
|
30
|
+
self.consumed = 0
|
|
31
|
+
self._by_phase: Dict[str, int] = {}
|
|
32
|
+
self._plan = tuple(budget_plan)
|
|
33
|
+
# per-phase soft caps (allocation of total) keyed by the 4 plan phases.
|
|
34
|
+
self._caps = {
|
|
35
|
+
phase: int(round(self.total * frac))
|
|
36
|
+
for phase, frac in zip(_BUDGET_PLAN_PHASES, self._plan)
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
def _plan_phase(self, phase: str) -> str:
|
|
40
|
+
if phase in _BUDGET_PLAN_PHASES:
|
|
41
|
+
return phase
|
|
42
|
+
if phase == "diagnose":
|
|
43
|
+
return "assess"
|
|
44
|
+
if phase in ("consolidate", "calibrate"):
|
|
45
|
+
return "update"
|
|
46
|
+
return "drill"
|
|
47
|
+
|
|
48
|
+
def charge(self, phase: str, n: int = 1) -> int:
|
|
49
|
+
if phase not in PRACTICE_PHASES and phase not in ("review", "promotion_sweep"):
|
|
50
|
+
raise ValueError(f"unknown budget phase {phase!r}")
|
|
51
|
+
if n < 0:
|
|
52
|
+
raise ValueError("charge n must be >= 0")
|
|
53
|
+
if self.consumed + n > self.total:
|
|
54
|
+
raise BudgetExhausted(
|
|
55
|
+
f"budget exhausted: consumed={self.consumed} + {n} > total={self.total}"
|
|
56
|
+
)
|
|
57
|
+
self.consumed += n
|
|
58
|
+
self._by_phase[phase] = self._by_phase.get(phase, 0) + n
|
|
59
|
+
return self.consumed
|
|
60
|
+
|
|
61
|
+
def remaining(self) -> int:
|
|
62
|
+
return self.total - self.consumed
|
|
63
|
+
|
|
64
|
+
def slice(self, phase: str, fraction: float) -> int:
|
|
65
|
+
"""Return an integer sub-budget handed to inner operators (their declared
|
|
66
|
+
eval_budget IS the slice). Bounded by remaining budget."""
|
|
67
|
+
if not 0.0 <= fraction <= 1.0:
|
|
68
|
+
raise ValueError("slice fraction must be in [0, 1]")
|
|
69
|
+
want = int(self.total * fraction)
|
|
70
|
+
return max(0, min(want, self.remaining()))
|
|
71
|
+
|
|
72
|
+
def ledger(self) -> dict:
|
|
73
|
+
"""Per-phase consumption; conservation: sum(phase) == consumed <= total."""
|
|
74
|
+
by_phase = {p: self._by_phase.get(p, 0) for p in sorted(self._by_phase)}
|
|
75
|
+
assert sum(by_phase.values()) == self.consumed <= self.total
|
|
76
|
+
return {
|
|
77
|
+
"total": self.total,
|
|
78
|
+
"consumed": self.consumed,
|
|
79
|
+
"remaining": self.remaining(),
|
|
80
|
+
"by_phase": by_phase,
|
|
81
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Unit 13 (BBG U13 / ARCH §2d phase 6) — CALIBRATE: the learned-gate.
|
|
2
|
+
|
|
3
|
+
Per cell: learned iff score ≥ floor AND fork-entropy ≤ threshold AND ICC ≥ floor
|
|
4
|
+
over k; high-score/high-entropy = fluent_not_learned (stays in rotation);
|
|
5
|
+
plateaued/zpd_exited stop rules. Trajectory profiles are post-hoc, never a stop
|
|
6
|
+
rule. Emits ``agent-learning.practice-calibration.v1``.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
11
|
+
|
|
12
|
+
from .._schema import public_payload
|
|
13
|
+
from ..live._contract import UNSTABLE_ICC_FLOOR
|
|
14
|
+
from ._contract import AGENT_LEARNING_PRACTICE_CALIBRATION_KIND, CALIBRATION_VERDICTS
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def calibrate_cell(
|
|
18
|
+
cell: Mapping[str, Any],
|
|
19
|
+
*,
|
|
20
|
+
score: float,
|
|
21
|
+
fork_entropy: float,
|
|
22
|
+
divergence_step: Optional[int],
|
|
23
|
+
icc: float,
|
|
24
|
+
repeats: int,
|
|
25
|
+
score_floor: float = 0.7,
|
|
26
|
+
entropy_threshold: float = 0.3,
|
|
27
|
+
icc_floor: float = UNSTABLE_ICC_FLOOR,
|
|
28
|
+
prior_score: Optional[float] = None,
|
|
29
|
+
in_band: bool = True,
|
|
30
|
+
) -> dict:
|
|
31
|
+
"""Compute one cell's calibration verdict (synthesis §4(6))."""
|
|
32
|
+
learned = score >= score_floor and fork_entropy <= entropy_threshold and icc >= icc_floor
|
|
33
|
+
if learned:
|
|
34
|
+
verdict = "learned"
|
|
35
|
+
stop_reason = "learned"
|
|
36
|
+
elif score >= score_floor and fork_entropy > entropy_threshold:
|
|
37
|
+
verdict = "fluent_not_learned" # high-score / high-entropy
|
|
38
|
+
stop_reason = None
|
|
39
|
+
elif not in_band:
|
|
40
|
+
verdict = "zpd_exited"
|
|
41
|
+
stop_reason = "zpd_exited"
|
|
42
|
+
elif prior_score is not None and abs(score - prior_score) < 1e-3:
|
|
43
|
+
verdict = "plateaued"
|
|
44
|
+
stop_reason = "plateaued"
|
|
45
|
+
else:
|
|
46
|
+
verdict = "in_rotation"
|
|
47
|
+
stop_reason = None
|
|
48
|
+
assert verdict in CALIBRATION_VERDICTS
|
|
49
|
+
return {
|
|
50
|
+
"cell": dict(cell),
|
|
51
|
+
"score": round(float(score), 6),
|
|
52
|
+
"fork_entropy": round(float(fork_entropy), 6),
|
|
53
|
+
"divergence_step": divergence_step,
|
|
54
|
+
"icc": round(float(icc), 6),
|
|
55
|
+
"repeats": int(repeats),
|
|
56
|
+
"verdict": verdict,
|
|
57
|
+
"stop_reason": stop_reason,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def calibrate(cells: Sequence[Mapping[str, Any]], *, round_no: int) -> dict:
|
|
62
|
+
"""Emit the calibration artifact over a list of pre-computed cell measures."""
|
|
63
|
+
records = [calibrate_cell(**c) if "verdict" not in c else dict(c) for c in cells]
|
|
64
|
+
report = {
|
|
65
|
+
"kind": AGENT_LEARNING_PRACTICE_CALIBRATION_KIND,
|
|
66
|
+
"round": int(round_no),
|
|
67
|
+
"cells": records,
|
|
68
|
+
}
|
|
69
|
+
return public_payload(report, kind=AGENT_LEARNING_PRACTICE_CALIBRATION_KIND)
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Unit 22 (BBG U22 / RU-7) — the capstone A/B harness.
|
|
2
|
+
|
|
3
|
+
An EXPERIMENT, not a release gate (gates stay deterministic; nothing here
|
|
4
|
+
registers a check). The harness runs the practice loop vs real search backends
|
|
5
|
+
at EQUAL TOTAL metered budget (the one meter, AD-I) over kit-local fixtures, and
|
|
6
|
+
REFUSES to print a headline unless every arm completed the same declared total
|
|
7
|
+
(``headline: null`` + ``ab_budget_mismatch`` otherwise — doctrine #11).
|
|
8
|
+
|
|
9
|
+
This module builds the harness so it CAN run offline-deterministically; running
|
|
10
|
+
the capstone experiment + writing the paper is a separate later task.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import List
|
|
17
|
+
|
|
18
|
+
from .._schema import public_payload
|
|
19
|
+
from ._contract import AGENT_LEARNING_PRACTICE_LOOP_KIND
|
|
20
|
+
|
|
21
|
+
# RU-7: real backend tokens only (the canon tuple stays closed). "greedy" = bandit.
|
|
22
|
+
CAPSTONE_ARMS = ("practice_loop", "gepa", "tpe", "society", "bandit")
|
|
23
|
+
# manifest-level ablation knobs of the practice arm (never a code fork).
|
|
24
|
+
CAPSTONE_ABLATIONS = ("a1_no_zpd", "a2_no_spacing", "a3_no_consolidation", "a4_no_calibration")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _load_config(manifest_dir: Path) -> dict:
|
|
28
|
+
config_path = manifest_dir / "capstone.json"
|
|
29
|
+
if not config_path.exists():
|
|
30
|
+
raise FileNotFoundError(f"capstone config not found at {config_path}")
|
|
31
|
+
return json.loads(config_path.read_text())
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def run_ab(manifest_dir: str | Path) -> dict:
|
|
35
|
+
"""Run the A/B harness. Reads ``capstone.json`` declaring the arms and the
|
|
36
|
+
equal total budget; enforces the equal-budget headline rule (doctrine #11).
|
|
37
|
+
|
|
38
|
+
The arm execution is offline-deterministic: each arm reports its declared
|
|
39
|
+
total metered budget and a (placeholder until the experiment runs)
|
|
40
|
+
retention_after_interference. Running the experiment itself is a later task;
|
|
41
|
+
this harness validates the equal-budget contract and emits the ab_harness
|
|
42
|
+
block."""
|
|
43
|
+
manifest_dir = Path(manifest_dir)
|
|
44
|
+
config = _load_config(manifest_dir)
|
|
45
|
+
declared_total = int(config.get("eval_budget", 0))
|
|
46
|
+
arms_decl = config.get("arms") or list(CAPSTONE_ARMS)
|
|
47
|
+
|
|
48
|
+
arms: List[dict] = []
|
|
49
|
+
budgets: set[int] = set()
|
|
50
|
+
for arm in arms_decl:
|
|
51
|
+
arm_total = int(config.get("arm_budgets", {}).get(arm, declared_total))
|
|
52
|
+
budgets.add(arm_total)
|
|
53
|
+
arms.append({
|
|
54
|
+
"arm": arm,
|
|
55
|
+
"total_metered_budget": arm_total,
|
|
56
|
+
# best_found is printed per arm precisely so a search arm may visibly
|
|
57
|
+
# win best-found while losing retention (the headline).
|
|
58
|
+
"best_found": None,
|
|
59
|
+
"retention_after_interference": None,
|
|
60
|
+
})
|
|
61
|
+
|
|
62
|
+
# equal TOTAL metered budget per arm (AD-I) — else headline null + warning.
|
|
63
|
+
budget_match = len(budgets) == 1 and declared_total in budgets
|
|
64
|
+
findings: List[dict] = []
|
|
65
|
+
headline = None
|
|
66
|
+
if not budget_match:
|
|
67
|
+
findings.append({
|
|
68
|
+
"type": "ab_budget_mismatch", "level": "warning",
|
|
69
|
+
"reason": f"arms did not complete the same declared total ({sorted(budgets)} != {declared_total})",
|
|
70
|
+
})
|
|
71
|
+
else:
|
|
72
|
+
headline = {"metric": "retention_after_interference", "by_arm": None,
|
|
73
|
+
"note": "populated when the experiment runs (a later task)"}
|
|
74
|
+
|
|
75
|
+
payload = {
|
|
76
|
+
"kind": AGENT_LEARNING_PRACTICE_LOOP_KIND,
|
|
77
|
+
"ab_harness": {
|
|
78
|
+
"arms": arms,
|
|
79
|
+
"ablations": list(CAPSTONE_ABLATIONS),
|
|
80
|
+
"equal_total_budget": declared_total,
|
|
81
|
+
"budget_match": budget_match,
|
|
82
|
+
"headline": headline,
|
|
83
|
+
"findings": findings,
|
|
84
|
+
},
|
|
85
|
+
}
|
|
86
|
+
return public_payload(payload, kind=AGENT_LEARNING_PRACTICE_LOOP_KIND)["ab_harness"]
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Unit 8 (BBG U8 / ARCH §3) — practice vocabularies + canon constants.
|
|
2
|
+
|
|
3
|
+
Every constant ARCH §3 freezes for the Practice Loop, verbatim. RU-1 numeric
|
|
4
|
+
defaults. Evidence/verdict vocab is IMPORTED from live/_contract.py, never
|
|
5
|
+
redeclared (ARCH §1.7). The Unit-20 gate byte-compares these.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from ..live._contract import ( # noqa: F401 (re-exported canon)
|
|
13
|
+
DEFAULT_REPEATS,
|
|
14
|
+
EVIDENCE_CLASSES,
|
|
15
|
+
RELEASE_ADMISSIBLE_EVIDENCE_CLASSES,
|
|
16
|
+
UNSTABLE_ICC_FLOOR,
|
|
17
|
+
VERDICTS,
|
|
18
|
+
)
|
|
19
|
+
from ..loss import ( # noqa: F401 (re-export the objective/loss-report kinds)
|
|
20
|
+
AGENT_LEARNING_LOSS_REPORT_KIND,
|
|
21
|
+
AGENT_LEARNING_OBJECTIVE_KIND,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
# --- artifact kinds (RU-4) -------------------------------------------------
|
|
25
|
+
AGENT_LEARNING_PRACTICE_LOOP_KIND = "agent-learning.practice-loop.v1"
|
|
26
|
+
AGENT_LEARNING_PRACTICE_RESULT_KIND = "agent-learning.practice-result.v1"
|
|
27
|
+
AGENT_LEARNING_PRACTICE_REPORT_KIND = "agent-learning.practice-report.v1"
|
|
28
|
+
AGENT_LEARNING_PRACTICE_DEFICITS_KIND = "agent-learning.practice-deficits.v1"
|
|
29
|
+
AGENT_LEARNING_PRACTICE_DRILL_KIND = "agent-learning.practice-drill.v1"
|
|
30
|
+
AGENT_LEARNING_PRACTICE_UPDATE_KIND = "agent-learning.practice-update.v1"
|
|
31
|
+
AGENT_LEARNING_CONSOLIDATED_LESSON_KIND = "agent-learning.consolidated-lesson.v1"
|
|
32
|
+
AGENT_LEARNING_PRACTICE_CALIBRATION_KIND = "agent-learning.practice-calibration.v1"
|
|
33
|
+
|
|
34
|
+
PRACTICE_ARTIFACT_KINDS = (
|
|
35
|
+
AGENT_LEARNING_PRACTICE_LOOP_KIND,
|
|
36
|
+
AGENT_LEARNING_PRACTICE_RESULT_KIND,
|
|
37
|
+
AGENT_LEARNING_PRACTICE_REPORT_KIND,
|
|
38
|
+
AGENT_LEARNING_PRACTICE_DEFICITS_KIND,
|
|
39
|
+
AGENT_LEARNING_PRACTICE_DRILL_KIND,
|
|
40
|
+
AGENT_LEARNING_PRACTICE_UPDATE_KIND,
|
|
41
|
+
AGENT_LEARNING_CONSOLIDATED_LESSON_KIND,
|
|
42
|
+
AGENT_LEARNING_PRACTICE_CALIBRATION_KIND,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
# --- phases + vocabularies (ARCH §3) ---------------------------------------
|
|
46
|
+
PRACTICE_PHASES = ("assess", "diagnose", "drill", "update", "consolidate", "calibrate")
|
|
47
|
+
SCAFFOLD_TYPES = ("world_simplification", "hint_tool", "worked_example", "relaxed_success")
|
|
48
|
+
ZPD_VERDICTS = ("in_band", "vygotsky_form", "below_band", "above_band", "unstable")
|
|
49
|
+
CALIBRATION_VERDICTS = ("learned", "fluent_not_learned", "in_rotation", "plateaued", "zpd_exited")
|
|
50
|
+
LADDER_STATES = ("episodic", "instruction", "skill")
|
|
51
|
+
PRACTICE_REPLAY_INTERVALS = (1, 2, 4, 8, 16) # cap 16
|
|
52
|
+
STORE_STATUSES = ("active", "retired")
|
|
53
|
+
RETIREMENT_REASONS = ("repeated_failure", "obsolete")
|
|
54
|
+
LESSON_KINDS = ("instruction_block", "config_patch", "skill")
|
|
55
|
+
|
|
56
|
+
# --- 13D-5 capstone ablation knobs (additive; the experiment path only) -----
|
|
57
|
+
# Real trainer config flags that change run_practice_loop behaviour (never
|
|
58
|
+
# labels): A1 disables ZPD filtering, A2 disables standing spaced reviews
|
|
59
|
+
# (replay only at promotion), A3 skips the consolidate phase entirely, A4
|
|
60
|
+
# disables the calibration learned-gate (fixed-k, never stop early).
|
|
61
|
+
PRACTICE_ABLATIONS = ("a1_no_zpd", "a2_no_spacing", "a3_no_consolidation", "a4_no_calibration")
|
|
62
|
+
|
|
63
|
+
# --- RU-1 defaults ---------------------------------------------------------
|
|
64
|
+
ZPD_BAND = (0.2, 0.7)
|
|
65
|
+
REVIEW_RATIO = 0.25
|
|
66
|
+
BUDGET_PLAN = (0.25, 0.35, 0.25, 0.15) # assess / drill / update / review
|
|
67
|
+
PRACTICE_STORE_ACTIVE_CAP = 64
|
|
68
|
+
SCAFFOLD_FADE_DEFAULT = (1.0, 0.5, 0.0) # MUST end at 0.0
|
|
69
|
+
MAX_REPLAY_INTERVAL = 16
|
|
70
|
+
DEFAULT_MAX_ROUNDS = 8
|
|
71
|
+
DEFAULT_INNER_OPERATOR_BACKEND = "society"
|
|
72
|
+
|
|
73
|
+
# --- store placement (AD-G — the Phase-8 ledger precedent) -----------------
|
|
74
|
+
LESSON_ID_PREFIX = "lesson_"
|
|
75
|
+
PRACTICE_STORE_PATH_ENV = "AGENT_LEARNING_PRACTICE_STORE_PATH"
|
|
76
|
+
PRACTICE_STORE_HOME_ENV = "AGENT_LEARNING_HOME"
|
|
77
|
+
PRACTICE_STORE_DIR_NAME = "practice"
|
|
78
|
+
PRACTICE_STORE_FILE_NAME = "records.jsonl"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def practice_store_path(override: str | Path | None = None) -> Path:
|
|
82
|
+
"""Resolve the consolidation store path (AD-G). Precedence: explicit arg >
|
|
83
|
+
AGENT_LEARNING_PRACTICE_STORE_PATH > ${AGENT_LEARNING_HOME:-~/.agent-learning}
|
|
84
|
+
/practice/records.jsonl."""
|
|
85
|
+
if override is not None:
|
|
86
|
+
return Path(override)
|
|
87
|
+
env_override = os.environ.get(PRACTICE_STORE_PATH_ENV)
|
|
88
|
+
if env_override:
|
|
89
|
+
return Path(env_override)
|
|
90
|
+
home = os.environ.get(PRACTICE_STORE_HOME_ENV) or (Path.home() / ".agent-learning")
|
|
91
|
+
return Path(home) / PRACTICE_STORE_DIR_NAME / PRACTICE_STORE_FILE_NAME
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Unit 10 (BBG U10 / ARCH §2d phase 2) — DIAGNOSE: pure composition.
|
|
2
|
+
|
|
3
|
+
Ranks weak cells, attributes each to a harness_layer ∈ HARNESS_LAYERS via
|
|
4
|
+
ComponentDiagnosis and relevant_search_paths (narrowing, never widening). Credit
|
|
5
|
+
method "layer_scoped" always; "counterfactual_replay" (13C T7) only budget-
|
|
6
|
+
permitting (fallback = layer scoping only). Emits
|
|
7
|
+
``agent-learning.practice-deficits.v1`` ranked deterministically (loss desc,
|
|
8
|
+
tie-break by cell content hash). No new machinery.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from typing import Any, List, Mapping, Optional
|
|
14
|
+
|
|
15
|
+
from .._schema import public_payload
|
|
16
|
+
from ._contract import AGENT_LEARNING_PRACTICE_DEFICITS_KIND
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _components():
|
|
20
|
+
import importlib
|
|
21
|
+
return importlib.import_module("fi.opt.components")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _cell_hash(cell: Mapping[str, Any]) -> str:
|
|
25
|
+
return json.dumps(cell, sort_keys=True, default=str)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def diagnose(
|
|
29
|
+
practice_report: Mapping[str, Any],
|
|
30
|
+
*,
|
|
31
|
+
search_space: Mapping[str, Any],
|
|
32
|
+
layer_hint: Optional[Mapping[str, str]] = None,
|
|
33
|
+
allow_counterfactual: bool = False,
|
|
34
|
+
) -> dict:
|
|
35
|
+
"""Pure composition over the ASSESS report's loss cells. ``layer_hint`` maps a
|
|
36
|
+
cell key → harness_layer (from upstream diagnosis); default 'execution'."""
|
|
37
|
+
components = _components()
|
|
38
|
+
harness_layers = components.HARNESS_LAYERS
|
|
39
|
+
prefixes = components.HARNESS_LAYER_PATH_PREFIXES
|
|
40
|
+
layer_hint = dict(layer_hint or {})
|
|
41
|
+
|
|
42
|
+
loss_report = practice_report.get("loss_report") or {}
|
|
43
|
+
cells = loss_report.get("cells") or []
|
|
44
|
+
# rank weak cells: loss desc, tie-break by cell content hash.
|
|
45
|
+
ranked = sorted(
|
|
46
|
+
cells,
|
|
47
|
+
key=lambda c: (-float(c.get("loss", 0.0)), _cell_hash(c.get("cell") or {})),
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
deficits: List[dict] = []
|
|
51
|
+
for cell_report in ranked:
|
|
52
|
+
cell = cell_report.get("cell") or {}
|
|
53
|
+
if float(cell_report.get("loss", 0.0)) <= 0.0:
|
|
54
|
+
continue # closed cells are not deficits
|
|
55
|
+
layer = layer_hint.get(_cell_hash(cell), "execution")
|
|
56
|
+
if layer not in harness_layers:
|
|
57
|
+
layer = "execution"
|
|
58
|
+
# narrowing search paths from the layer's prefixes.
|
|
59
|
+
layer_prefixes = prefixes.get(layer, ())
|
|
60
|
+
narrowed = sorted(
|
|
61
|
+
path for path in search_space
|
|
62
|
+
if any(path == p or path.startswith(f"{p}.") for p in layer_prefixes)
|
|
63
|
+
)
|
|
64
|
+
method = "counterfactual_replay" if allow_counterfactual else "layer_scoped"
|
|
65
|
+
deficits.append({
|
|
66
|
+
"cell": cell,
|
|
67
|
+
"harness_layer": layer,
|
|
68
|
+
"search_paths": narrowed,
|
|
69
|
+
"credit": {"method": method, "rows": []},
|
|
70
|
+
"evidence_rows": cell_report.get("verdicts") or [],
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
report = {
|
|
74
|
+
"kind": AGENT_LEARNING_PRACTICE_DEFICITS_KIND,
|
|
75
|
+
"round": practice_report.get("round"),
|
|
76
|
+
"objective_version": practice_report.get("objective_version"),
|
|
77
|
+
"deficits": deficits,
|
|
78
|
+
}
|
|
79
|
+
return public_payload(report, kind=AGENT_LEARNING_PRACTICE_DEFICITS_KIND)
|