agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,665 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Kubernetes backend for cloud-native job execution.
|
|
3
|
+
|
|
4
|
+
Provides distributed task execution using Kubernetes Jobs.
|
|
5
|
+
Each task is submitted as a K8s Job that runs a container which
|
|
6
|
+
deserializes and executes the function, then writes JSON results to stdout.
|
|
7
|
+
|
|
8
|
+
Requires: pip install kubernetes cloudpickle
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
import re
|
|
13
|
+
import threading
|
|
14
|
+
import time
|
|
15
|
+
import uuid
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from typing import Any, Callable, Dict, List, Optional, TypeVar
|
|
18
|
+
|
|
19
|
+
from .base import Backend, BackendConfig, TaskHandle, TaskStatus
|
|
20
|
+
from ._utils import KUBERNETES
|
|
21
|
+
from ._container import (
|
|
22
|
+
EVAL_PAYLOAD_ENV,
|
|
23
|
+
RUNNER_COMMAND,
|
|
24
|
+
parse_result_from_logs,
|
|
25
|
+
serialize_task,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
T = TypeVar("T")
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class KubernetesConfig(BackendConfig):
|
|
34
|
+
"""
|
|
35
|
+
Configuration for Kubernetes backend.
|
|
36
|
+
|
|
37
|
+
A pre-built eval runner Dockerfile is provided at
|
|
38
|
+
``fi/evals/framework/backends/Dockerfile.eval-runner``.
|
|
39
|
+
Build it with::
|
|
40
|
+
|
|
41
|
+
docker build -f Dockerfile.eval-runner -t fi-eval-runner:latest .
|
|
42
|
+
|
|
43
|
+
Then pass ``image="fi-eval-runner:latest"`` (or your registry path).
|
|
44
|
+
|
|
45
|
+
Attributes:
|
|
46
|
+
namespace: Kubernetes namespace for Jobs
|
|
47
|
+
image: Container image (must have cloudpickle installed)
|
|
48
|
+
cpu_request: CPU request per Job pod
|
|
49
|
+
cpu_limit: CPU limit per Job pod
|
|
50
|
+
memory_request: Memory request per Job pod
|
|
51
|
+
memory_limit: Memory limit per Job pod
|
|
52
|
+
job_prefix: Prefix for Job names (DNS-1123 compliant)
|
|
53
|
+
backoff_limit: Number of retries for failed Jobs
|
|
54
|
+
active_deadline_seconds: Maximum seconds a Job may run
|
|
55
|
+
ttl_seconds_after_finished: Seconds to keep finished Jobs before cleanup
|
|
56
|
+
service_account_name: K8s service account for the Job pod
|
|
57
|
+
image_pull_secrets: Names of image pull secrets
|
|
58
|
+
image_pull_policy: Image pull policy (Always, IfNotPresent, Never)
|
|
59
|
+
kubeconfig_path: Path to kubeconfig file (None = auto-detect)
|
|
60
|
+
context: Kubeconfig context to use (None = current context)
|
|
61
|
+
in_cluster: Force in-cluster config loading
|
|
62
|
+
labels: Extra labels to apply to Jobs
|
|
63
|
+
annotations: Extra annotations to apply to Jobs
|
|
64
|
+
poll_interval: Seconds between status polls
|
|
65
|
+
log_tail_lines: Number of log lines to tail (None = all)
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
namespace: str = "default"
|
|
69
|
+
image: str = "fi-eval-runner:latest"
|
|
70
|
+
cpu_request: str = "500m"
|
|
71
|
+
cpu_limit: str = "1"
|
|
72
|
+
memory_request: str = "512Mi"
|
|
73
|
+
memory_limit: str = "2Gi"
|
|
74
|
+
job_prefix: str = "eval-"
|
|
75
|
+
backoff_limit: int = 0
|
|
76
|
+
active_deadline_seconds: int = 600
|
|
77
|
+
ttl_seconds_after_finished: int = 300
|
|
78
|
+
service_account_name: Optional[str] = None
|
|
79
|
+
image_pull_secrets: Optional[List[str]] = None
|
|
80
|
+
image_pull_policy: str = "IfNotPresent"
|
|
81
|
+
kubeconfig_path: Optional[str] = None
|
|
82
|
+
context: Optional[str] = None
|
|
83
|
+
in_cluster: Optional[bool] = None
|
|
84
|
+
labels: Optional[Dict[str, str]] = None
|
|
85
|
+
annotations: Optional[Dict[str, str]] = None
|
|
86
|
+
poll_interval: float = 2.0
|
|
87
|
+
log_tail_lines: Optional[int] = None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class KubernetesBackend(Backend):
|
|
91
|
+
"""
|
|
92
|
+
Kubernetes backend for cloud-native job execution.
|
|
93
|
+
|
|
94
|
+
Submits evaluation tasks as Kubernetes Jobs. Each task is serialized
|
|
95
|
+
with cloudpickle, base64-encoded, and passed to the container as an env var.
|
|
96
|
+
Results are read from pod logs as JSON.
|
|
97
|
+
|
|
98
|
+
Example:
|
|
99
|
+
config = KubernetesConfig(
|
|
100
|
+
namespace="evaluations",
|
|
101
|
+
image="my-eval-image:latest",
|
|
102
|
+
cpu_request="1",
|
|
103
|
+
memory_limit="4Gi",
|
|
104
|
+
)
|
|
105
|
+
backend = KubernetesBackend(config)
|
|
106
|
+
|
|
107
|
+
handle = backend.submit(my_eval_func, args=(input_data,))
|
|
108
|
+
result = backend.get_result(handle)
|
|
109
|
+
|
|
110
|
+
# Or batch submission
|
|
111
|
+
handles = backend.submit_batch([
|
|
112
|
+
(eval_func, (data1,), {}, None),
|
|
113
|
+
(eval_func, (data2,), {}, None),
|
|
114
|
+
])
|
|
115
|
+
|
|
116
|
+
Note:
|
|
117
|
+
Requires a running Kubernetes cluster. Uses in-cluster config
|
|
118
|
+
when running inside a pod, or falls back to kubeconfig for local dev.
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
name: str = "kubernetes"
|
|
122
|
+
|
|
123
|
+
def __init__(self, config: Optional[KubernetesConfig] = None):
|
|
124
|
+
"""
|
|
125
|
+
Initialize Kubernetes backend.
|
|
126
|
+
|
|
127
|
+
Args:
|
|
128
|
+
config: Kubernetes configuration
|
|
129
|
+
|
|
130
|
+
Raises:
|
|
131
|
+
ImportError: If kubernetes package is not installed
|
|
132
|
+
"""
|
|
133
|
+
KUBERNETES.require()
|
|
134
|
+
|
|
135
|
+
self.config = config or KubernetesConfig()
|
|
136
|
+
self._batch_api: Optional[Any] = None
|
|
137
|
+
self._core_api: Optional[Any] = None
|
|
138
|
+
self._handles: Dict[str, TaskHandle] = {}
|
|
139
|
+
self._job_names: Dict[str, str] = {}
|
|
140
|
+
self._lock = threading.Lock()
|
|
141
|
+
|
|
142
|
+
self._setup_kubernetes()
|
|
143
|
+
|
|
144
|
+
def _setup_kubernetes(self) -> None:
|
|
145
|
+
"""Load kubeconfig and create API clients."""
|
|
146
|
+
from kubernetes import client, config
|
|
147
|
+
|
|
148
|
+
if self.config.in_cluster is True:
|
|
149
|
+
config.load_incluster_config()
|
|
150
|
+
elif self.config.in_cluster is False:
|
|
151
|
+
config.load_kube_config(
|
|
152
|
+
config_file=self.config.kubeconfig_path,
|
|
153
|
+
context=self.config.context,
|
|
154
|
+
)
|
|
155
|
+
else:
|
|
156
|
+
# Auto-detect: try in-cluster first, fall back to kubeconfig
|
|
157
|
+
try:
|
|
158
|
+
config.load_incluster_config()
|
|
159
|
+
except config.ConfigException:
|
|
160
|
+
config.load_kube_config(
|
|
161
|
+
config_file=self.config.kubeconfig_path,
|
|
162
|
+
context=self.config.context,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
self._batch_api = client.BatchV1Api()
|
|
166
|
+
self._core_api = client.CoreV1Api()
|
|
167
|
+
logger.info(
|
|
168
|
+
f"Kubernetes backend initialized (namespace={self.config.namespace})"
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
def _make_job_name(self, task_id: str) -> str:
|
|
172
|
+
"""
|
|
173
|
+
Generate a DNS-1123 compliant Job name.
|
|
174
|
+
|
|
175
|
+
The name is ``{job_prefix}{uuid_hex[:12]}``, lowercased and
|
|
176
|
+
stripped of any characters that are not lowercase alphanumeric
|
|
177
|
+
or hyphens. The total length is capped at 63 characters.
|
|
178
|
+
"""
|
|
179
|
+
short_id = uuid.uuid4().hex[:12]
|
|
180
|
+
raw = f"{self.config.job_prefix}{short_id}"
|
|
181
|
+
# DNS-1123: lowercase, alphanumeric and hyphens only
|
|
182
|
+
name = re.sub(r"[^a-z0-9-]", "", raw.lower())
|
|
183
|
+
# Must start/end with alphanumeric
|
|
184
|
+
name = name.strip("-")
|
|
185
|
+
return name[:63]
|
|
186
|
+
|
|
187
|
+
def _build_job_manifest(
|
|
188
|
+
self, job_name: str, serialized_payload: str
|
|
189
|
+
) -> Any:
|
|
190
|
+
"""Build a V1Job manifest for the given payload."""
|
|
191
|
+
from kubernetes import client
|
|
192
|
+
|
|
193
|
+
labels = {"app": "fi-eval", "job-name": job_name}
|
|
194
|
+
if self.config.labels:
|
|
195
|
+
labels.update(self.config.labels)
|
|
196
|
+
|
|
197
|
+
annotations = self.config.annotations or {}
|
|
198
|
+
|
|
199
|
+
# Resource requirements
|
|
200
|
+
resources = client.V1ResourceRequirements(
|
|
201
|
+
requests={
|
|
202
|
+
"cpu": self.config.cpu_request,
|
|
203
|
+
"memory": self.config.memory_request,
|
|
204
|
+
},
|
|
205
|
+
limits={
|
|
206
|
+
"cpu": self.config.cpu_limit,
|
|
207
|
+
"memory": self.config.memory_limit,
|
|
208
|
+
},
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
container = client.V1Container(
|
|
212
|
+
name="eval-runner",
|
|
213
|
+
image=self.config.image,
|
|
214
|
+
image_pull_policy=self.config.image_pull_policy,
|
|
215
|
+
command=RUNNER_COMMAND,
|
|
216
|
+
env=[
|
|
217
|
+
client.V1EnvVar(
|
|
218
|
+
name=EVAL_PAYLOAD_ENV, value=serialized_payload
|
|
219
|
+
),
|
|
220
|
+
],
|
|
221
|
+
resources=resources,
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
# Image pull secrets
|
|
225
|
+
image_pull_secrets = None
|
|
226
|
+
if self.config.image_pull_secrets:
|
|
227
|
+
image_pull_secrets = [
|
|
228
|
+
client.V1LocalObjectReference(name=s)
|
|
229
|
+
for s in self.config.image_pull_secrets
|
|
230
|
+
]
|
|
231
|
+
|
|
232
|
+
pod_spec = client.V1PodSpec(
|
|
233
|
+
containers=[container],
|
|
234
|
+
restart_policy="Never",
|
|
235
|
+
service_account_name=self.config.service_account_name,
|
|
236
|
+
image_pull_secrets=image_pull_secrets,
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
template = client.V1PodTemplateSpec(
|
|
240
|
+
metadata=client.V1ObjectMeta(labels=labels, annotations=annotations),
|
|
241
|
+
spec=pod_spec,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
job_spec = client.V1JobSpec(
|
|
245
|
+
template=template,
|
|
246
|
+
backoff_limit=self.config.backoff_limit,
|
|
247
|
+
active_deadline_seconds=self.config.active_deadline_seconds,
|
|
248
|
+
ttl_seconds_after_finished=self.config.ttl_seconds_after_finished,
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
job = client.V1Job(
|
|
252
|
+
api_version="batch/v1",
|
|
253
|
+
kind="Job",
|
|
254
|
+
metadata=client.V1ObjectMeta(
|
|
255
|
+
name=job_name,
|
|
256
|
+
namespace=self.config.namespace,
|
|
257
|
+
labels=labels,
|
|
258
|
+
annotations=annotations,
|
|
259
|
+
),
|
|
260
|
+
spec=job_spec,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
return job
|
|
264
|
+
|
|
265
|
+
def submit(
|
|
266
|
+
self,
|
|
267
|
+
fn: Callable[..., T],
|
|
268
|
+
args: tuple = (),
|
|
269
|
+
kwargs: Optional[Dict[str, Any]] = None,
|
|
270
|
+
context: Optional[Dict[str, Any]] = None,
|
|
271
|
+
) -> TaskHandle[T]:
|
|
272
|
+
"""
|
|
273
|
+
Submit a task as a Kubernetes Job.
|
|
274
|
+
|
|
275
|
+
The function and its arguments are serialized with cloudpickle,
|
|
276
|
+
base64-encoded, and passed to the container as an env var.
|
|
277
|
+
|
|
278
|
+
Args:
|
|
279
|
+
fn: The function to execute
|
|
280
|
+
args: Positional arguments
|
|
281
|
+
kwargs: Keyword arguments
|
|
282
|
+
context: Trace context (stored in metadata)
|
|
283
|
+
|
|
284
|
+
Returns:
|
|
285
|
+
TaskHandle to track the task
|
|
286
|
+
"""
|
|
287
|
+
kwargs = kwargs or {}
|
|
288
|
+
task_id = str(uuid.uuid4())
|
|
289
|
+
job_name = self._make_job_name(task_id)
|
|
290
|
+
|
|
291
|
+
handle = TaskHandle(
|
|
292
|
+
task_id=task_id,
|
|
293
|
+
backend_name=self.name,
|
|
294
|
+
metadata={
|
|
295
|
+
"function": fn.__name__ if hasattr(fn, "__name__") else str(fn),
|
|
296
|
+
"context": context,
|
|
297
|
+
"job_name": job_name,
|
|
298
|
+
"namespace": self.config.namespace,
|
|
299
|
+
},
|
|
300
|
+
)
|
|
301
|
+
handle._status = TaskStatus.PENDING
|
|
302
|
+
|
|
303
|
+
with self._lock:
|
|
304
|
+
self._handles[task_id] = handle
|
|
305
|
+
self._job_names[task_id] = job_name
|
|
306
|
+
|
|
307
|
+
try:
|
|
308
|
+
serialized = serialize_task(fn, args, kwargs)
|
|
309
|
+
|
|
310
|
+
# Build and create the Job
|
|
311
|
+
job_manifest = self._build_job_manifest(job_name, serialized)
|
|
312
|
+
self._batch_api.create_namespaced_job(
|
|
313
|
+
namespace=self.config.namespace,
|
|
314
|
+
body=job_manifest,
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
with self._lock:
|
|
318
|
+
handle._status = TaskStatus.RUNNING
|
|
319
|
+
|
|
320
|
+
logger.debug(
|
|
321
|
+
f"Submitted task {task_id} as Job {job_name} "
|
|
322
|
+
f"in namespace {self.config.namespace}"
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
except Exception as e:
|
|
326
|
+
handle._status = TaskStatus.FAILED
|
|
327
|
+
handle._error = str(e)
|
|
328
|
+
logger.error(f"Failed to submit task {task_id}: {e}")
|
|
329
|
+
|
|
330
|
+
return handle
|
|
331
|
+
|
|
332
|
+
def get_result(
|
|
333
|
+
self,
|
|
334
|
+
handle: TaskHandle[T],
|
|
335
|
+
timeout: Optional[float] = None,
|
|
336
|
+
) -> T:
|
|
337
|
+
"""
|
|
338
|
+
Get the result of a Kubernetes Job by polling until completion.
|
|
339
|
+
|
|
340
|
+
Args:
|
|
341
|
+
handle: The task handle from submit()
|
|
342
|
+
timeout: Maximum seconds to wait
|
|
343
|
+
|
|
344
|
+
Returns:
|
|
345
|
+
The task result
|
|
346
|
+
|
|
347
|
+
Raises:
|
|
348
|
+
TimeoutError: If timeout exceeded
|
|
349
|
+
RuntimeError: If the Job failed
|
|
350
|
+
ValueError: If task is unknown
|
|
351
|
+
"""
|
|
352
|
+
timeout = timeout or self.config.timeout_seconds
|
|
353
|
+
|
|
354
|
+
with self._lock:
|
|
355
|
+
job_name = self._job_names.get(handle.task_id)
|
|
356
|
+
|
|
357
|
+
if job_name is None:
|
|
358
|
+
raise ValueError(f"Unknown task: {handle.task_id}")
|
|
359
|
+
|
|
360
|
+
deadline = time.monotonic() + timeout
|
|
361
|
+
while True:
|
|
362
|
+
status = self._poll_job_status(job_name)
|
|
363
|
+
|
|
364
|
+
if status == TaskStatus.COMPLETED:
|
|
365
|
+
result = self._read_job_result(job_name)
|
|
366
|
+
with self._lock:
|
|
367
|
+
if handle.task_id in self._handles:
|
|
368
|
+
self._handles[handle.task_id]._status = TaskStatus.COMPLETED
|
|
369
|
+
self._handles[handle.task_id]._result = result
|
|
370
|
+
return result
|
|
371
|
+
|
|
372
|
+
if status == TaskStatus.FAILED:
|
|
373
|
+
error_msg = f"Job {job_name} failed"
|
|
374
|
+
# Try to read logs for error details
|
|
375
|
+
try:
|
|
376
|
+
self._read_job_result(job_name)
|
|
377
|
+
except RuntimeError as re_err:
|
|
378
|
+
error_msg = str(re_err)
|
|
379
|
+
except Exception:
|
|
380
|
+
pass
|
|
381
|
+
with self._lock:
|
|
382
|
+
if handle.task_id in self._handles:
|
|
383
|
+
self._handles[handle.task_id]._status = TaskStatus.FAILED
|
|
384
|
+
self._handles[handle.task_id]._error = error_msg
|
|
385
|
+
raise RuntimeError(error_msg)
|
|
386
|
+
|
|
387
|
+
if time.monotonic() > deadline:
|
|
388
|
+
with self._lock:
|
|
389
|
+
if handle.task_id in self._handles:
|
|
390
|
+
self._handles[handle.task_id]._status = TaskStatus.TIMEOUT
|
|
391
|
+
raise TimeoutError(
|
|
392
|
+
f"Task {handle.task_id} timed out after {timeout}s"
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
time.sleep(self.config.poll_interval)
|
|
396
|
+
|
|
397
|
+
def _poll_job_status(self, job_name: str) -> TaskStatus:
|
|
398
|
+
"""Poll the K8s API for the Job's current status."""
|
|
399
|
+
job = self._batch_api.read_namespaced_job_status(
|
|
400
|
+
name=job_name,
|
|
401
|
+
namespace=self.config.namespace,
|
|
402
|
+
)
|
|
403
|
+
return self._map_job_status(job.status)
|
|
404
|
+
|
|
405
|
+
def _map_job_status(self, status: Any) -> TaskStatus:
|
|
406
|
+
"""Map a V1JobStatus object to TaskStatus."""
|
|
407
|
+
# Check conditions first (Complete / Failed)
|
|
408
|
+
if status.conditions:
|
|
409
|
+
for condition in status.conditions:
|
|
410
|
+
if condition.type == "Complete" and condition.status == "True":
|
|
411
|
+
return TaskStatus.COMPLETED
|
|
412
|
+
if condition.type == "Failed" and condition.status == "True":
|
|
413
|
+
return TaskStatus.FAILED
|
|
414
|
+
|
|
415
|
+
# If pods are still active, the job is running
|
|
416
|
+
if status.active and status.active > 0:
|
|
417
|
+
return TaskStatus.RUNNING
|
|
418
|
+
|
|
419
|
+
# Succeeded / failed counts
|
|
420
|
+
if status.succeeded and status.succeeded > 0:
|
|
421
|
+
return TaskStatus.COMPLETED
|
|
422
|
+
if status.failed and status.failed > 0:
|
|
423
|
+
return TaskStatus.FAILED
|
|
424
|
+
|
|
425
|
+
return TaskStatus.PENDING
|
|
426
|
+
|
|
427
|
+
def _read_job_result(self, job_name: str) -> Any:
|
|
428
|
+
"""
|
|
429
|
+
Read the result from the Job's pod logs.
|
|
430
|
+
|
|
431
|
+
Uses :func:`_container.parse_result_from_logs` to extract the
|
|
432
|
+
JSON result printed by the runner script.
|
|
433
|
+
"""
|
|
434
|
+
from kubernetes.client.rest import ApiException
|
|
435
|
+
|
|
436
|
+
# Find pods belonging to this Job
|
|
437
|
+
label_selector = f"job-name={job_name}"
|
|
438
|
+
pods = self._core_api.list_namespaced_pod(
|
|
439
|
+
namespace=self.config.namespace,
|
|
440
|
+
label_selector=label_selector,
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
if not pods.items:
|
|
444
|
+
raise RuntimeError(f"No pods found for Job {job_name}")
|
|
445
|
+
|
|
446
|
+
pod_name = pods.items[0].metadata.name
|
|
447
|
+
|
|
448
|
+
log_kwargs = {"name": pod_name, "namespace": self.config.namespace}
|
|
449
|
+
if self.config.log_tail_lines is not None:
|
|
450
|
+
log_kwargs["tail_lines"] = self.config.log_tail_lines
|
|
451
|
+
|
|
452
|
+
try:
|
|
453
|
+
# _preload_content=False prevents the K8s client from
|
|
454
|
+
# auto-deserializing JSON log lines into Python objects
|
|
455
|
+
resp = self._core_api.read_namespaced_pod_log(
|
|
456
|
+
**log_kwargs, _preload_content=False
|
|
457
|
+
)
|
|
458
|
+
logs = resp.data.decode("utf-8")
|
|
459
|
+
except ApiException as e:
|
|
460
|
+
raise RuntimeError(f"Failed to read logs for pod {pod_name}: {e}")
|
|
461
|
+
|
|
462
|
+
return parse_result_from_logs(logs)
|
|
463
|
+
|
|
464
|
+
def get_status(self, handle: TaskHandle) -> TaskStatus:
|
|
465
|
+
"""
|
|
466
|
+
Get current status of a Kubernetes Job.
|
|
467
|
+
|
|
468
|
+
Returns cached terminal status if available, otherwise polls K8s.
|
|
469
|
+
|
|
470
|
+
Args:
|
|
471
|
+
handle: The task handle
|
|
472
|
+
|
|
473
|
+
Returns:
|
|
474
|
+
Current TaskStatus
|
|
475
|
+
"""
|
|
476
|
+
with self._lock:
|
|
477
|
+
if handle.task_id in self._handles:
|
|
478
|
+
cached = self._handles[handle.task_id]._status
|
|
479
|
+
if cached in (
|
|
480
|
+
TaskStatus.COMPLETED,
|
|
481
|
+
TaskStatus.FAILED,
|
|
482
|
+
TaskStatus.CANCELLED,
|
|
483
|
+
TaskStatus.TIMEOUT,
|
|
484
|
+
):
|
|
485
|
+
return cached
|
|
486
|
+
|
|
487
|
+
job_name = self._job_names.get(handle.task_id)
|
|
488
|
+
|
|
489
|
+
if job_name is None:
|
|
490
|
+
return TaskStatus.FAILED
|
|
491
|
+
|
|
492
|
+
try:
|
|
493
|
+
status = self._poll_job_status(job_name)
|
|
494
|
+
with self._lock:
|
|
495
|
+
if handle.task_id in self._handles:
|
|
496
|
+
self._handles[handle.task_id]._status = status
|
|
497
|
+
return status
|
|
498
|
+
except Exception:
|
|
499
|
+
return TaskStatus.FAILED
|
|
500
|
+
|
|
501
|
+
def cancel(self, handle: TaskHandle) -> bool:
|
|
502
|
+
"""
|
|
503
|
+
Cancel a Kubernetes Job.
|
|
504
|
+
|
|
505
|
+
Deletes the Job with foreground propagation policy so
|
|
506
|
+
associated pods are terminated immediately.
|
|
507
|
+
|
|
508
|
+
Args:
|
|
509
|
+
handle: The task handle
|
|
510
|
+
|
|
511
|
+
Returns:
|
|
512
|
+
True if cancelled, False otherwise
|
|
513
|
+
"""
|
|
514
|
+
from kubernetes.client.rest import ApiException
|
|
515
|
+
|
|
516
|
+
with self._lock:
|
|
517
|
+
job_name = self._job_names.get(handle.task_id)
|
|
518
|
+
|
|
519
|
+
if job_name is None:
|
|
520
|
+
return False
|
|
521
|
+
|
|
522
|
+
try:
|
|
523
|
+
from kubernetes import client
|
|
524
|
+
|
|
525
|
+
self._batch_api.delete_namespaced_job(
|
|
526
|
+
name=job_name,
|
|
527
|
+
namespace=self.config.namespace,
|
|
528
|
+
body=client.V1DeleteOptions(
|
|
529
|
+
propagation_policy="Foreground",
|
|
530
|
+
),
|
|
531
|
+
)
|
|
532
|
+
|
|
533
|
+
with self._lock:
|
|
534
|
+
if handle.task_id in self._handles:
|
|
535
|
+
self._handles[handle.task_id]._status = TaskStatus.CANCELLED
|
|
536
|
+
|
|
537
|
+
logger.debug(f"Cancelled Job {job_name}")
|
|
538
|
+
return True
|
|
539
|
+
|
|
540
|
+
except ApiException as e:
|
|
541
|
+
logger.warning(f"Failed to cancel Job {job_name}: {e}")
|
|
542
|
+
return False
|
|
543
|
+
|
|
544
|
+
def submit_batch(
|
|
545
|
+
self,
|
|
546
|
+
tasks: List[tuple],
|
|
547
|
+
) -> List[TaskHandle]:
|
|
548
|
+
"""
|
|
549
|
+
Submit multiple tasks as Kubernetes Jobs.
|
|
550
|
+
|
|
551
|
+
Submits tasks sequentially (each creates an independent K8s Job).
|
|
552
|
+
|
|
553
|
+
Args:
|
|
554
|
+
tasks: List of (function, args, kwargs, context) tuples
|
|
555
|
+
|
|
556
|
+
Returns:
|
|
557
|
+
List of TaskHandles
|
|
558
|
+
"""
|
|
559
|
+
handles = []
|
|
560
|
+
for fn, args, kwargs, context in tasks:
|
|
561
|
+
handle = self.submit(fn, args, kwargs or {}, context)
|
|
562
|
+
handles.append(handle)
|
|
563
|
+
return handles
|
|
564
|
+
|
|
565
|
+
def shutdown(self, wait: bool = True) -> None:
|
|
566
|
+
"""
|
|
567
|
+
Shutdown the Kubernetes backend.
|
|
568
|
+
|
|
569
|
+
Args:
|
|
570
|
+
wait: If False, cancels all pending/running tasks
|
|
571
|
+
"""
|
|
572
|
+
if not wait:
|
|
573
|
+
with self._lock:
|
|
574
|
+
pending_tasks = [
|
|
575
|
+
(tid, jn)
|
|
576
|
+
for tid, jn in self._job_names.items()
|
|
577
|
+
if tid in self._handles
|
|
578
|
+
and self._handles[tid]._status
|
|
579
|
+
in (TaskStatus.PENDING, TaskStatus.RUNNING)
|
|
580
|
+
]
|
|
581
|
+
|
|
582
|
+
for task_id, _job_name in pending_tasks:
|
|
583
|
+
try:
|
|
584
|
+
handle = TaskHandle(task_id=task_id, backend_name=self.name)
|
|
585
|
+
self.cancel(handle)
|
|
586
|
+
except Exception:
|
|
587
|
+
pass
|
|
588
|
+
|
|
589
|
+
logger.info("Kubernetes backend shut down")
|
|
590
|
+
|
|
591
|
+
def get_job_logs(self, handle: TaskHandle) -> Dict[str, str]:
|
|
592
|
+
"""
|
|
593
|
+
Get raw logs for all pods of a Job.
|
|
594
|
+
|
|
595
|
+
Args:
|
|
596
|
+
handle: The task handle
|
|
597
|
+
|
|
598
|
+
Returns:
|
|
599
|
+
Dict mapping pod name to log text
|
|
600
|
+
|
|
601
|
+
Raises:
|
|
602
|
+
ValueError: If task is unknown
|
|
603
|
+
"""
|
|
604
|
+
from kubernetes.client.rest import ApiException
|
|
605
|
+
|
|
606
|
+
with self._lock:
|
|
607
|
+
job_name = self._job_names.get(handle.task_id)
|
|
608
|
+
|
|
609
|
+
if job_name is None:
|
|
610
|
+
raise ValueError(f"Unknown task: {handle.task_id}")
|
|
611
|
+
|
|
612
|
+
label_selector = f"job-name={job_name}"
|
|
613
|
+
pods = self._core_api.list_namespaced_pod(
|
|
614
|
+
namespace=self.config.namespace,
|
|
615
|
+
label_selector=label_selector,
|
|
616
|
+
)
|
|
617
|
+
|
|
618
|
+
result: Dict[str, str] = {}
|
|
619
|
+
for pod in pods.items:
|
|
620
|
+
pod_name = pod.metadata.name
|
|
621
|
+
try:
|
|
622
|
+
logs = self._core_api.read_namespaced_pod_log(
|
|
623
|
+
name=pod_name,
|
|
624
|
+
namespace=self.config.namespace,
|
|
625
|
+
)
|
|
626
|
+
result[pod_name] = logs
|
|
627
|
+
except ApiException as e:
|
|
628
|
+
result[pod_name] = f"<error reading logs: {e}>"
|
|
629
|
+
|
|
630
|
+
return result
|
|
631
|
+
|
|
632
|
+
def get_stats(self) -> dict:
|
|
633
|
+
"""
|
|
634
|
+
Get Kubernetes backend statistics.
|
|
635
|
+
|
|
636
|
+
Returns:
|
|
637
|
+
Dictionary with task counts and config info
|
|
638
|
+
"""
|
|
639
|
+
with self._lock:
|
|
640
|
+
pending = len([
|
|
641
|
+
h for h in self._handles.values()
|
|
642
|
+
if h._status == TaskStatus.PENDING
|
|
643
|
+
])
|
|
644
|
+
running = len([
|
|
645
|
+
h for h in self._handles.values()
|
|
646
|
+
if h._status == TaskStatus.RUNNING
|
|
647
|
+
])
|
|
648
|
+
completed = len([
|
|
649
|
+
h for h in self._handles.values()
|
|
650
|
+
if h._status == TaskStatus.COMPLETED
|
|
651
|
+
])
|
|
652
|
+
failed = len([
|
|
653
|
+
h for h in self._handles.values()
|
|
654
|
+
if h._status == TaskStatus.FAILED
|
|
655
|
+
])
|
|
656
|
+
|
|
657
|
+
return {
|
|
658
|
+
"namespace": self.config.namespace,
|
|
659
|
+
"image": self.config.image,
|
|
660
|
+
"pending_tasks": pending,
|
|
661
|
+
"running_tasks": running,
|
|
662
|
+
"completed_tasks": completed,
|
|
663
|
+
"failed_tasks": failed,
|
|
664
|
+
"total_tasks": len(self._handles),
|
|
665
|
+
}
|