agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: agent-learning-kit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Unified Future AGI SDK for agent learning workflows.
|
|
5
|
+
Project-URL: Homepage, https://futureagi.com
|
|
6
|
+
Project-URL: Documentation, https://docs.futureagi.com
|
|
7
|
+
Project-URL: Repository, https://github.com/future-agi/agent-learning-kit
|
|
8
|
+
Project-URL: Issues, https://github.com/future-agi/agent-learning-kit/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/future-agi/agent-learning-kit/blob/main/CHANGELOG.md
|
|
10
|
+
Author-email: Future AGI <hello@futureagi.io>
|
|
11
|
+
License: Apache-2.0
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
License-File: NOTICE
|
|
14
|
+
Keywords: agent-evaluation,agent-optimization,agent-simulation,agent-testing,ai-agents,future-agi,red-teaming
|
|
15
|
+
Classifier: Development Status :: 4 - Beta
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Software Development :: Testing
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Requires-Dist: claude-agent-sdk>=0.2.139
|
|
27
|
+
Requires-Dist: fi-instrumentation-otel>=0.1.16
|
|
28
|
+
Requires-Dist: gepa>=0.0.17
|
|
29
|
+
Requires-Dist: google-adk<2.8,>=2.7.1
|
|
30
|
+
Requires-Dist: httpx>=0.24.0
|
|
31
|
+
Requires-Dist: jsonschema<5,>=4.25.1
|
|
32
|
+
Requires-Dist: levenshtein>=0.25.0
|
|
33
|
+
Requires-Dist: litellm<2,>=1.80.0
|
|
34
|
+
Requires-Dist: nltk>=3.9.0
|
|
35
|
+
Requires-Dist: numpy>=1.26.4
|
|
36
|
+
Requires-Dist: openai<3,>=1.109.1
|
|
37
|
+
Requires-Dist: opentelemetry-api<2,>=1.39.1
|
|
38
|
+
Requires-Dist: opentelemetry-exporter-otlp<2,>=1.39.1
|
|
39
|
+
Requires-Dist: opentelemetry-sdk<2,>=1.39.1
|
|
40
|
+
Requires-Dist: optuna>=3.6.1
|
|
41
|
+
Requires-Dist: pandas>=2.0.0
|
|
42
|
+
Requires-Dist: pydantic<3,>=2.0
|
|
43
|
+
Requires-Dist: python-dotenv>=1.0.0
|
|
44
|
+
Requires-Dist: pyyaml>=6.0
|
|
45
|
+
Requires-Dist: requests-futures>=1.0.0
|
|
46
|
+
Requires-Dist: requests<3,>=2.32.5
|
|
47
|
+
Requires-Dist: retell-sdk<6,>=5.64
|
|
48
|
+
Requires-Dist: rich>=13.0.0
|
|
49
|
+
Requires-Dist: rouge-score>=0.1.2
|
|
50
|
+
Requires-Dist: typer<1.0.0,>=0.9.0
|
|
51
|
+
Provides-Extra: a2a
|
|
52
|
+
Requires-Dist: a2a-sdk[http-server]>=1.1.0; extra == 'a2a'
|
|
53
|
+
Provides-Extra: all
|
|
54
|
+
Requires-Dist: aiohttp>=3.10; extra == 'all'
|
|
55
|
+
Requires-Dist: audioop-lts>=0.2.1; (python_version >= '3.13') and extra == 'all'
|
|
56
|
+
Requires-Dist: chromadb>=0.4.0; extra == 'all'
|
|
57
|
+
Requires-Dist: livekit-agents[cartesia,deepgram,google,openai,silero]>=1.2; extra == 'all'
|
|
58
|
+
Requires-Dist: livekit-plugins-elevenlabs>=1.2; extra == 'all'
|
|
59
|
+
Requires-Dist: sentence-transformers<6,>=5.2.3; extra == 'all'
|
|
60
|
+
Requires-Dist: torch<3,>=2.10.0; extra == 'all'
|
|
61
|
+
Requires-Dist: transformers<6,>=5.2.0; extra == 'all'
|
|
62
|
+
Provides-Extra: embeddings
|
|
63
|
+
Requires-Dist: sentence-transformers<6,>=5.2.3; extra == 'embeddings'
|
|
64
|
+
Provides-Extra: evaluation
|
|
65
|
+
Provides-Extra: feedback
|
|
66
|
+
Requires-Dist: chromadb>=0.4.0; extra == 'feedback'
|
|
67
|
+
Provides-Extra: harness-stores
|
|
68
|
+
Requires-Dist: psycopg[binary]>=3.2; extra == 'harness-stores'
|
|
69
|
+
Provides-Extra: harness-ui
|
|
70
|
+
Requires-Dist: fastapi<1,>=0.115; extra == 'harness-ui'
|
|
71
|
+
Requires-Dist: uvicorn<1,>=0.30; extra == 'harness-ui'
|
|
72
|
+
Provides-Extra: langchain
|
|
73
|
+
Requires-Dist: langchain-core<2,>=1.4.6; extra == 'langchain'
|
|
74
|
+
Requires-Dist: langgraph-checkpoint-sqlite>=3.1.0; extra == 'langchain'
|
|
75
|
+
Requires-Dist: langgraph<2,>=1.2.4; extra == 'langchain'
|
|
76
|
+
Provides-Extra: livekit
|
|
77
|
+
Requires-Dist: aiohttp>=3.10; extra == 'livekit'
|
|
78
|
+
Requires-Dist: audioop-lts>=0.2.1; (python_version >= '3.13') and extra == 'livekit'
|
|
79
|
+
Requires-Dist: livekit-agents[cartesia,deepgram,google,openai,silero]>=1.2; extra == 'livekit'
|
|
80
|
+
Requires-Dist: livekit-plugins-elevenlabs>=1.2; extra == 'livekit'
|
|
81
|
+
Provides-Extra: mcp
|
|
82
|
+
Requires-Dist: mcp<2,>=1.27; extra == 'mcp'
|
|
83
|
+
Provides-Extra: nli
|
|
84
|
+
Requires-Dist: torch<3,>=2.10.0; extra == 'nli'
|
|
85
|
+
Requires-Dist: transformers<6,>=5.2.0; extra == 'nli'
|
|
86
|
+
Provides-Extra: notebook
|
|
87
|
+
Requires-Dist: ipykernel>=6; extra == 'notebook'
|
|
88
|
+
Requires-Dist: nbformat>=5; extra == 'notebook'
|
|
89
|
+
Provides-Extra: optimize
|
|
90
|
+
Provides-Extra: pipecat
|
|
91
|
+
Requires-Dist: pipecat-ai>=0.0.108; extra == 'pipecat'
|
|
92
|
+
Provides-Extra: simulate
|
|
93
|
+
Provides-Extra: trinity
|
|
94
|
+
Requires-Dist: aiohttp>=3.10; extra == 'trinity'
|
|
95
|
+
Requires-Dist: audioop-lts>=0.2.1; (python_version >= '3.13') and extra == 'trinity'
|
|
96
|
+
Requires-Dist: livekit-agents[cartesia,deepgram,google,openai,silero]>=1.2; extra == 'trinity'
|
|
97
|
+
Requires-Dist: livekit-plugins-elevenlabs>=1.2; extra == 'trinity'
|
|
98
|
+
Description-Content-Type: text/markdown
|
|
99
|
+
|
|
100
|
+
<p align="center">
|
|
101
|
+
<img src="https://raw.githubusercontent.com/future-agi/agent-learning-kit/main/docs/assets/futureagi-mark-email.png" alt="Future AGI" width="72" />
|
|
102
|
+
</p>
|
|
103
|
+
|
|
104
|
+
<h1 align="center">Agent Learning Kit</h1>
|
|
105
|
+
|
|
106
|
+
<p align="center">
|
|
107
|
+
Local-first testing, simulation, red teaming, and optimization for AI agents.
|
|
108
|
+
</p>
|
|
109
|
+
|
|
110
|
+
<p align="center">
|
|
111
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/LICENSE">Apache-2.0</a>
|
|
112
|
+
·
|
|
113
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/docs/index.md">Docs</a>
|
|
114
|
+
·
|
|
115
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/CONTRIBUTING.md">Contributing</a>
|
|
116
|
+
·
|
|
117
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/SECURITY.md">Security</a>
|
|
118
|
+
·
|
|
119
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/ROADMAP.md">V1 roadmap</a>
|
|
120
|
+
·
|
|
121
|
+
<a href="https://github.com/future-agi/agent-learning-kit/blob/main/LIBRARIES.md">Library inventory</a>
|
|
122
|
+
</p>
|
|
123
|
+
|
|
124
|
+

|
|
125
|
+
|
|
126
|
+
Agent Learning Kit is the local-first SDK and CLI for testing, simulating,
|
|
127
|
+
red-teaming, and optimizing AI agents.
|
|
128
|
+
|
|
129
|
+
It brings the three core Future AGI engines into one public developer surface —
|
|
130
|
+
three engines, four workflows: red-teaming rides on the `simulate` and `evals`
|
|
131
|
+
engines rather than being a fourth engine:
|
|
132
|
+
|
|
133
|
+
- `simulate`: run local worlds, tasks, framework-shaped adapters, replays, and
|
|
134
|
+
regression artifacts.
|
|
135
|
+
- `evals`: evaluate prompts, task outputs, runtime contracts, traces, memory,
|
|
136
|
+
retrieval, safety, and robustness evidence.
|
|
137
|
+
- `optimize`: search over prompts, agents, framework adapters, worlds,
|
|
138
|
+
multi-agent interactions, memory layers, workflows, and red-team scenarios.
|
|
139
|
+
|
|
140
|
+
Use it when you want one reproducible loop:
|
|
141
|
+
|
|
142
|
+
1. Simulate an agent or framework workflow.
|
|
143
|
+
2. Evaluate the behavior and runtime evidence.
|
|
144
|
+
3. Optimize the weak layer.
|
|
145
|
+
4. Promote the result into a replayable artifact.
|
|
146
|
+
5. Prove release readiness with local gates.
|
|
147
|
+
|
|
148
|
+
### The harness: point it at an agent and talk to it
|
|
149
|
+
|
|
150
|
+
`src/fi/alk/harness/` builds all of the above **for** an agent instead of asking you to write it.
|
|
151
|
+
Point it at an agent's source and it reads what that agent verifiably is, builds a real world its
|
|
152
|
+
tools act on, and writes test scenarios that are each proved before they are kept. It is driven
|
|
153
|
+
as a conversation, in a terminal or on a web page.
|
|
154
|
+
|
|
155
|
+
- **[Start here](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/alk/harness/README.md)**: setup from nothing, then how to use it
|
|
156
|
+
- **[The web page](https://github.com/future-agi/agent-learning-kit/blob/main/harness-ui/README.md)**: the same harness as a chat, on `localhost:8777`
|
|
157
|
+
- **[How it works](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/alk/harness/HOW-IT-WORKS.md)** and
|
|
158
|
+
**[why it is shaped this way](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/alk/harness/DESIGN.md)**
|
|
159
|
+
|
|
160
|
+
OpenEnv/Gymnasium shapes are compatibility inputs, not the product center.
|
|
161
|
+
Agent Learning Kit is the primary runtime and release contract, and the bar is
|
|
162
|
+
the executable `environment_10x_robustness` release gate.
|
|
163
|
+
OpenEnv/Gymnasium-shaped traces remain compatibility evidence inside that bar.
|
|
164
|
+
|
|
165
|
+
## Install
|
|
166
|
+
|
|
167
|
+
Install from PyPI:
|
|
168
|
+
|
|
169
|
+
```bash
|
|
170
|
+
pip install agent-learning-kit
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
To develop against source (contributors):
|
|
174
|
+
|
|
175
|
+
```bash
|
|
176
|
+
git clone https://github.com/future-agi/agent-learning-kit
|
|
177
|
+
cd agent-learning-kit
|
|
178
|
+
uv sync # or: pip install -e .
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
(npm publishing of the TypeScript SDK lands at the v1 launch.)
|
|
182
|
+
|
|
183
|
+
Optional Python extras:
|
|
184
|
+
|
|
185
|
+
```bash
|
|
186
|
+
pip install "agent-learning-kit[livekit]"
|
|
187
|
+
pip install "agent-learning-kit[nli]"
|
|
188
|
+
pip install "agent-learning-kit[all]"
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
TypeScript evaluation package (npm at launch; today build from
|
|
192
|
+
[`typescript/agent-learning-kit`](https://github.com/future-agi/agent-learning-kit/blob/main/typescript/agent-learning-kit)):
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
pnpm add @future-agi/agent-learning-kit
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## Quickstart
|
|
199
|
+
|
|
200
|
+
Everything below runs fully offline — no API key, no network. Start with the
|
|
201
|
+
local doctor:
|
|
202
|
+
|
|
203
|
+
```bash
|
|
204
|
+
agent-learn doctor
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
Then run the golden path against the bundled example manifests. The
|
|
208
|
+
`AGENT_LEARNING_*_EXAMPLE_KEY` prefixes satisfy each manifest's
|
|
209
|
+
`required_env` list — that list is CI wiring metadata, not a provider
|
|
210
|
+
credential, so any placeholder value works.
|
|
211
|
+
|
|
212
|
+
> Prefer the SDK spine over the CLI?
|
|
213
|
+
> [Spec + Runner](https://github.com/future-agi/agent-learning-kit/blob/main/docs/simulate/spec-and-runner.md) runs the same simulation as
|
|
214
|
+
> one `SimulationSpec` fed to one `SimulationRunner` — the plug-and-play surface
|
|
215
|
+
> behind every simulation.
|
|
216
|
+
|
|
217
|
+
Evaluate a suite:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
agent-learn eval examples/eval_suite.json \
|
|
221
|
+
--output artifacts/eval.json
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
Simulate a run manifest:
|
|
225
|
+
|
|
226
|
+
```bash
|
|
227
|
+
AGENT_LEARNING_RUN_EXAMPLE_KEY=offline-demo-key \
|
|
228
|
+
agent-learn run examples/run_manifest.json \
|
|
229
|
+
--no-eval \
|
|
230
|
+
--output artifacts/run.json
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
Optimize an agent workflow:
|
|
234
|
+
|
|
235
|
+
```bash
|
|
236
|
+
AGENT_LEARNING_OPTIMIZE_EXAMPLE_KEY=offline-demo-key \
|
|
237
|
+
agent-learn optimize examples/optimization_manifest.json \
|
|
238
|
+
--output artifacts/optimization.json
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
Run a red-team campaign:
|
|
242
|
+
|
|
243
|
+
```bash
|
|
244
|
+
AGENT_LEARNING_REDTEAM_EXAMPLE_KEY=offline-demo-key \
|
|
245
|
+
agent-learn redteam examples/redteam_manifest.json \
|
|
246
|
+
--output artifacts/redteam.json
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
Each command prints a `wrote <path>` line; relative `--output` paths resolve
|
|
250
|
+
against your current working directory.
|
|
251
|
+
|
|
252
|
+
Optional platform mode: to use Future AGI platform-backed evaluation, set
|
|
253
|
+
`AGENT_LEARNING_API_KEY` (it takes precedence over the `FUTURE_AGI_API_KEY`
|
|
254
|
+
and `FI_API_KEY` aliases), or call `configure(api_key="...")` from
|
|
255
|
+
`fi.alk`. See
|
|
256
|
+
[docs/reference/configure.md](https://github.com/future-agi/agent-learning-kit/blob/main/docs/reference/configure.md).
|
|
257
|
+
|
|
258
|
+
Cut local release proof:
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
agent-learn release-check --project-root .
|
|
262
|
+
agent-learn release-proof \
|
|
263
|
+
--project-root . \
|
|
264
|
+
--output /tmp/agent-learning-release-proof.json \
|
|
265
|
+
--quiet
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
## TypeScript
|
|
269
|
+
|
|
270
|
+
```typescript
|
|
271
|
+
import { Evaluator } from "@future-agi/agent-learning-kit";
|
|
272
|
+
import { LocalEvaluator } from "@future-agi/agent-learning-kit/evals/local";
|
|
273
|
+
```
|
|
274
|
+
|
|
275
|
+
## What You Can Build
|
|
276
|
+
|
|
277
|
+
- Prompt and response evaluations.
|
|
278
|
+
- Local task and world simulations.
|
|
279
|
+
- Framework adapter probes (probe-promoted coverage) for LangChain, LangGraph,
|
|
280
|
+
LlamaIndex, AutoGen, CrewAI, LiveKit, Pipecat, Browser Use, MCP, A2A, and
|
|
281
|
+
custom orchestration objects.
|
|
282
|
+
- Runtime-simulated coverage for PydanticAI (multi-framework runtime
|
|
283
|
+
simulation) and OpenAI Agents (handoff-transcript promotion).
|
|
284
|
+
- Runtime-contract and trace-quality checks.
|
|
285
|
+
- Multi-agent coordination and handoff tests.
|
|
286
|
+
- Retrieval and memory quality checks.
|
|
287
|
+
- Voice, realtime, browser/CUA, workflow, lifecycle, and protocol traces.
|
|
288
|
+
- Red-team corpus, campaign, adaptive-loop, and persistent-state checks.
|
|
289
|
+
- Optimizer governance, candidate lineage, rollback, and release proof.
|
|
290
|
+
|
|
291
|
+
## Why It Exists
|
|
292
|
+
|
|
293
|
+
Most agent stacks split testing, simulation, optimization, and safety review
|
|
294
|
+
across separate tools. Agent Learning Kit keeps those steps in one artifact
|
|
295
|
+
model so a developer can inspect what happened, score it, improve it, and replay
|
|
296
|
+
it in CI.
|
|
297
|
+
|
|
298
|
+
The public SDK is `agent-learning-kit`, the Python namespace is
|
|
299
|
+
`fi.alk`, the CLI is `agent-learn`, and the TypeScript package is
|
|
300
|
+
`@future-agi/agent-learning-kit`.
|
|
301
|
+
|
|
302
|
+
The active `ai-evaluation` code is included here under `src/fi/evals`, with its
|
|
303
|
+
TypeScript SDK source under `typescript/agent-learning-kit/src`. The
|
|
304
|
+
`simulate-sdk` and `agent-opt` engine code is included under `src/fi/simulate`
|
|
305
|
+
and `src/fi/opt`. See [LIBRARIES.md](https://github.com/future-agi/agent-learning-kit/blob/main/LIBRARIES.md) for the complete source map.
|
|
306
|
+
The ai-evaluation source inventory used by `agent-learn release-check` lives at
|
|
307
|
+
the ai-evaluation source inventory (maintained in the internal-docs repo).
|
|
308
|
+
|
|
309
|
+
## Repository Map
|
|
310
|
+
|
|
311
|
+
- [`examples/`](https://github.com/future-agi/agent-learning-kit/blob/main/examples): runnable cookbooks and manifests.
|
|
312
|
+
- [`src/fi/alk`](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/alk): public Python SDK facade and CLI.
|
|
313
|
+
- [`src/fi/evals`](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/evals): active `ai-evaluation` engine code.
|
|
314
|
+
- [`src/fi/simulate`](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/simulate): migrated `simulate-sdk` engine code.
|
|
315
|
+
- [`src/fi/opt`](https://github.com/future-agi/agent-learning-kit/blob/main/src/fi/opt): migrated `agent-opt` engine code.
|
|
316
|
+
- [`typescript/agent-learning-kit`](https://github.com/future-agi/agent-learning-kit/blob/main/typescript/agent-learning-kit): public
|
|
317
|
+
TypeScript package, including the active evaluation SDK source.
|
|
318
|
+
- [`docs/index.md`](https://github.com/future-agi/agent-learning-kit/blob/main/docs/index.md): full documentation index.
|
|
319
|
+
- [`ROADMAP.md`](https://github.com/future-agi/agent-learning-kit/blob/main/ROADMAP.md): public v1 roadmap and post-v1 extensions.
|
|
320
|
+
- [`LIBRARIES.md`](https://github.com/future-agi/agent-learning-kit/blob/main/LIBRARIES.md): source map for the consolidated engines.
|
|
321
|
+
- [`CONTRIBUTING.md`](https://github.com/future-agi/agent-learning-kit/blob/main/CONTRIBUTING.md): local development and PR workflow.
|
|
322
|
+
- [`SECURITY.md`](https://github.com/future-agi/agent-learning-kit/blob/main/SECURITY.md): vulnerability reporting policy.
|
|
323
|
+
- [`LICENSE`](https://github.com/future-agi/agent-learning-kit/blob/main/LICENSE): Apache-2.0 license.
|
|
324
|
+
- [`NOTICE`](https://github.com/future-agi/agent-learning-kit/blob/main/NOTICE): Apache notice metadata.
|
|
325
|
+
|
|
326
|
+
## Development
|
|
327
|
+
|
|
328
|
+
New public SDK development belongs here. See [DEVELOPMENT.md](https://github.com/future-agi/agent-learning-kit/blob/main/DEVELOPMENT.md)
|
|
329
|
+
for the boundary between this package and the backing engine repos.
|
|
330
|
+
|
|
331
|
+
```bash
|
|
332
|
+
uv sync
|
|
333
|
+
uv run ruff check .
|
|
334
|
+
uv run pytest -q
|
|
335
|
+
uv run python -m build
|
|
336
|
+
pnpm --dir typescript --filter @future-agi/agent-learning-kit build
|
|
337
|
+
pnpm --dir typescript --filter @future-agi/agent-learning-kit test -- --runInBand
|
|
338
|
+
```
|
|
339
|
+
|
|
340
|
+
For the heavier release cut, run `agent-learn release-proof --project-root .`.
|
|
341
|
+
It emits `agent-learning.release-proof.v1` with command evidence for the full
|
|
342
|
+
local proof stack.
|
|
343
|
+
|
|
344
|
+
Before a release:
|
|
345
|
+
|
|
346
|
+
```bash
|
|
347
|
+
uv run python -m fi.alk.cli release-proof \
|
|
348
|
+
--project-root . \
|
|
349
|
+
--output /tmp/agent-learning-release-proof.json \
|
|
350
|
+
--quiet
|
|
351
|
+
```
|
|
352
|
+
|
|
353
|
+
`release-proof` includes release-check, full-repo ruff, pytest, Python package
|
|
354
|
+
build, TypeScript package build/test, and `git diff --check`. Use
|
|
355
|
+
`--only <check>` for partial proof during development or `--dry-run` to emit the
|
|
356
|
+
exact command plan without executing commands.
|
|
357
|
+
|
|
358
|
+
## Project Status
|
|
359
|
+
|
|
360
|
+
The v1 release gate is local-first and executable. It covers SDK consolidation,
|
|
361
|
+
promptfoo-style CLI usage, native optimizer evidence, docs/examples, schema
|
|
362
|
+
kinds, packaging metadata, red-team corpus/campaign coverage, Future AGI
|
|
363
|
+
UI/action/report artifacts, framework/provider compatibility, environment
|
|
364
|
+
robustness, regression replay, and release proof.
|
|
365
|
+
|
|
366
|
+
All v1 gates are green on the proved release commit (see the release-proof
|
|
367
|
+
artifact). Roadmap milestones marked "mostly complete" or "in progress" are
|
|
368
|
+
extend-only: the v1 contract those gates assert is frozen and proved; the named
|
|
369
|
+
extensions land post-v1 without weakening any gate.
|
|
370
|
+
|
|
371
|
+
## Community
|
|
372
|
+
|
|
373
|
+
- Contributions: [CONTRIBUTING.md](https://github.com/future-agi/agent-learning-kit/blob/main/CONTRIBUTING.md)
|
|
374
|
+
- Code of conduct: [CODE_OF_CONDUCT.md](https://github.com/future-agi/agent-learning-kit/blob/main/CODE_OF_CONDUCT.md)
|
|
375
|
+
- Security reports: [SECURITY.md](https://github.com/future-agi/agent-learning-kit/blob/main/SECURITY.md)
|
|
376
|
+
- License: [Apache-2.0](https://github.com/future-agi/agent-learning-kit/blob/main/LICENSE)
|
|
377
|
+
|
|
378
|
+
## Deep Dive
|
|
379
|
+
|
|
380
|
+
The full documentation set — quickstarts, per-track guides, framework pages,
|
|
381
|
+
and reference material — starts at [docs/index.md](https://github.com/future-agi/agent-learning-kit/blob/main/docs/index.md).
|