agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/suite.py
ADDED
|
@@ -0,0 +1,4200 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import copy
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import shlex
|
|
9
|
+
import sys
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
14
|
+
from xml.sax.saxutils import escape
|
|
15
|
+
|
|
16
|
+
from ._schema import AGENT_LEARNING_CLI_SCHEMA_VERSION, public_payload
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
AGENT_LEARNING_SUITE_KIND = "agent-learning.suite.v1"
|
|
20
|
+
AGENT_LEARNING_SUITE_OPTIMIZATION_KIND = "agent-learning.suite-optimization.v1"
|
|
21
|
+
AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND = (
|
|
22
|
+
"agent-learning.optimization-lifecycle.v1"
|
|
23
|
+
)
|
|
24
|
+
AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND = (
|
|
25
|
+
"agent-learning.suite.trust-certificate.v1"
|
|
26
|
+
)
|
|
27
|
+
AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND = (
|
|
28
|
+
"agent-learning.suite.trust-verification.v1"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
_CHILD_COMMANDS = {
|
|
32
|
+
"action_run",
|
|
33
|
+
"baseline",
|
|
34
|
+
"compare",
|
|
35
|
+
"promote_to_regression",
|
|
36
|
+
"replay",
|
|
37
|
+
"report",
|
|
38
|
+
"run",
|
|
39
|
+
"shrink",
|
|
40
|
+
"suite",
|
|
41
|
+
"eval",
|
|
42
|
+
"eval_artifact",
|
|
43
|
+
"eval_task",
|
|
44
|
+
"redteam",
|
|
45
|
+
"optimize",
|
|
46
|
+
"optimize_eval",
|
|
47
|
+
"optimize_suite",
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
_ADMITTED_EVIDENCE_ROLES = {
|
|
51
|
+
"admitted",
|
|
52
|
+
"claim",
|
|
53
|
+
"primary",
|
|
54
|
+
"paper_facing",
|
|
55
|
+
"paper_facing_evidence",
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
_NON_ADMITTED_EVIDENCE_ROLES = {
|
|
59
|
+
"calibration",
|
|
60
|
+
"diagnostic",
|
|
61
|
+
"fixture",
|
|
62
|
+
"preflight",
|
|
63
|
+
"smoke",
|
|
64
|
+
"support",
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class SuiteError(ValueError):
|
|
69
|
+
"""Raised when an Agent Learning suite manifest cannot run."""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class SuiteRunOptions:
|
|
74
|
+
name: Optional[str] = None
|
|
75
|
+
threshold: Optional[float] = None
|
|
76
|
+
max_candidates: Optional[int] = None
|
|
77
|
+
dry_run: bool = False
|
|
78
|
+
fail_fast: bool = False
|
|
79
|
+
require_optimizer_governance: bool = False
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class SuiteOptimizationOptions:
|
|
84
|
+
name: Optional[str] = None
|
|
85
|
+
threshold: Optional[float] = None
|
|
86
|
+
max_candidates: Optional[int] = None
|
|
87
|
+
dry_run: bool = False
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def load_suite_file(path: str | Path) -> dict[str, Any]:
|
|
91
|
+
suite_path = Path(path).expanduser().resolve()
|
|
92
|
+
if not suite_path.exists():
|
|
93
|
+
raise SuiteError(f"suite manifest not found: {suite_path}")
|
|
94
|
+
suite = _load_json_or_yaml(suite_path)
|
|
95
|
+
if not isinstance(suite, Mapping):
|
|
96
|
+
raise SuiteError("suite manifest root must be an object")
|
|
97
|
+
return _prepare_suite(dict(suite), base_dir=suite_path.parent)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def load_suite_artifact_file(path: str | Path) -> dict[str, Any]:
|
|
101
|
+
artifact_path = Path(path).expanduser().resolve()
|
|
102
|
+
if not artifact_path.exists():
|
|
103
|
+
raise SuiteError(f"suite artifact not found: {artifact_path}")
|
|
104
|
+
artifact = _load_json_or_yaml(artifact_path)
|
|
105
|
+
if not isinstance(artifact, Mapping):
|
|
106
|
+
raise SuiteError("suite artifact root must be an object")
|
|
107
|
+
return dict(artifact)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def verify_trust_certificate_file(
|
|
111
|
+
path: str | Path,
|
|
112
|
+
*,
|
|
113
|
+
required_verdict: str = "approved",
|
|
114
|
+
require_promotion_ready: bool = True,
|
|
115
|
+
) -> dict[str, Any]:
|
|
116
|
+
artifact_path = Path(path).expanduser().resolve()
|
|
117
|
+
artifact = load_suite_artifact_file(artifact_path)
|
|
118
|
+
return verify_trust_certificate(
|
|
119
|
+
artifact,
|
|
120
|
+
required_verdict=required_verdict,
|
|
121
|
+
require_promotion_ready=require_promotion_ready,
|
|
122
|
+
source_path=artifact_path,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def verify_trust_certificate(
|
|
127
|
+
artifact: Mapping[str, Any],
|
|
128
|
+
*,
|
|
129
|
+
required_verdict: str = "approved",
|
|
130
|
+
require_promotion_ready: bool = True,
|
|
131
|
+
source_path: str | Path | None = None,
|
|
132
|
+
) -> dict[str, Any]:
|
|
133
|
+
"""Verify a saved suite trust certificate without re-running the suite."""
|
|
134
|
+
required = _suite_key(required_verdict)
|
|
135
|
+
if required not in _TRUST_VERDICT_RANK:
|
|
136
|
+
allowed = ", ".join(sorted(_TRUST_VERDICT_RANK))
|
|
137
|
+
raise SuiteError(f"required_verdict must be one of: {allowed}")
|
|
138
|
+
|
|
139
|
+
source = Path(source_path).expanduser().resolve() if source_path else None
|
|
140
|
+
result_kind = str(artifact.get("kind") or artifact.get("version") or "")
|
|
141
|
+
summary = _as_mapping(artifact.get("summary"))
|
|
142
|
+
certificate = _as_mapping(artifact.get("trust_certificate"))
|
|
143
|
+
if not certificate and result_kind == AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND:
|
|
144
|
+
certificate = dict(artifact)
|
|
145
|
+
|
|
146
|
+
findings: list[dict[str, Any]] = []
|
|
147
|
+
certificate_kind = str(certificate.get("kind") or "") if certificate else ""
|
|
148
|
+
if not certificate:
|
|
149
|
+
findings.append({
|
|
150
|
+
"type": "suite_trust_certificate_missing",
|
|
151
|
+
"level": "error",
|
|
152
|
+
"reason": "Suite artifact does not contain a trust_certificate block.",
|
|
153
|
+
})
|
|
154
|
+
elif certificate_kind != AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND:
|
|
155
|
+
findings.append({
|
|
156
|
+
"type": "suite_trust_certificate_kind_mismatch",
|
|
157
|
+
"level": "error",
|
|
158
|
+
"reason": (
|
|
159
|
+
"Suite trust certificate kind must be "
|
|
160
|
+
f"{AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND}."
|
|
161
|
+
),
|
|
162
|
+
"observed_kind": certificate_kind,
|
|
163
|
+
})
|
|
164
|
+
|
|
165
|
+
observed = _suite_key(
|
|
166
|
+
certificate.get("verdict") if certificate else None
|
|
167
|
+
) or _suite_key(summary.get("trust_certificate_verdict"))
|
|
168
|
+
verdict_rank_passed = False
|
|
169
|
+
if certificate:
|
|
170
|
+
if observed not in _TRUST_VERDICT_RANK:
|
|
171
|
+
findings.append({
|
|
172
|
+
"type": "suite_trust_certificate_verdict_unknown",
|
|
173
|
+
"level": "error",
|
|
174
|
+
"reason": "Suite trust certificate verdict is missing or unknown.",
|
|
175
|
+
"observed_verdict": observed or None,
|
|
176
|
+
})
|
|
177
|
+
else:
|
|
178
|
+
verdict_rank_passed = (
|
|
179
|
+
_TRUST_VERDICT_RANK[observed] >= _TRUST_VERDICT_RANK[required]
|
|
180
|
+
)
|
|
181
|
+
if not verdict_rank_passed:
|
|
182
|
+
findings.append({
|
|
183
|
+
"type": "suite_trust_certificate_verdict_too_low",
|
|
184
|
+
"level": "error",
|
|
185
|
+
"reason": (
|
|
186
|
+
f"Suite trust certificate verdict {observed} is below "
|
|
187
|
+
f"required verdict {required}."
|
|
188
|
+
),
|
|
189
|
+
"required_verdict": required,
|
|
190
|
+
"observed_verdict": observed,
|
|
191
|
+
})
|
|
192
|
+
|
|
193
|
+
promotion_ready = _optional_bool(
|
|
194
|
+
certificate.get("promotion_ready") if certificate else None,
|
|
195
|
+
summary.get("trust_certificate_promotion_ready"),
|
|
196
|
+
)
|
|
197
|
+
promotion_gate_passed = not require_promotion_ready or promotion_ready is True
|
|
198
|
+
if certificate and not promotion_gate_passed:
|
|
199
|
+
findings.append({
|
|
200
|
+
"type": "suite_trust_certificate_not_promotion_ready",
|
|
201
|
+
"level": "error",
|
|
202
|
+
"reason": "Suite trust certificate is not marked promotion_ready.",
|
|
203
|
+
"promotion_ready": promotion_ready,
|
|
204
|
+
})
|
|
205
|
+
|
|
206
|
+
passed = not findings
|
|
207
|
+
return {
|
|
208
|
+
"kind": AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND,
|
|
209
|
+
"version": AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND,
|
|
210
|
+
"status": "passed" if passed else "failed",
|
|
211
|
+
"exit_code": 0 if passed else 1,
|
|
212
|
+
"source_path": str(source) if source else None,
|
|
213
|
+
"result_kind": result_kind or None,
|
|
214
|
+
"required_verdict": required,
|
|
215
|
+
"require_promotion_ready": bool(require_promotion_ready),
|
|
216
|
+
"observed_verdict": observed or None,
|
|
217
|
+
"promotion_ready": promotion_ready,
|
|
218
|
+
"certificate_kind": certificate_kind or None,
|
|
219
|
+
"assurance_level": (
|
|
220
|
+
certificate.get("assurance_level") if certificate else None
|
|
221
|
+
),
|
|
222
|
+
"summary": {
|
|
223
|
+
"certificate_present": bool(certificate),
|
|
224
|
+
"certificate_kind_passed": (
|
|
225
|
+
certificate_kind == AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND
|
|
226
|
+
),
|
|
227
|
+
"verdict_rank_passed": verdict_rank_passed,
|
|
228
|
+
"promotion_gate_passed": promotion_gate_passed,
|
|
229
|
+
"finding_count": len(findings),
|
|
230
|
+
},
|
|
231
|
+
"trust_certificate": copy.deepcopy(certificate),
|
|
232
|
+
"findings": findings,
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
load_suite = load_suite_file
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def required_suite_env(
|
|
240
|
+
suite: Mapping[str, Any],
|
|
241
|
+
*,
|
|
242
|
+
suite_path: str | Path = ".",
|
|
243
|
+
) -> list[str]:
|
|
244
|
+
base_dir = _suite_base_dir(suite_path)
|
|
245
|
+
required = set(_as_string_list(suite.get("required_env")))
|
|
246
|
+
for job in _suite_jobs(suite):
|
|
247
|
+
try:
|
|
248
|
+
child = _load_child_source(job, base_dir=base_dir)
|
|
249
|
+
except Exception:
|
|
250
|
+
continue
|
|
251
|
+
if _normalize_command(job.get("command") or job.get("type")) == "suite":
|
|
252
|
+
required.update(
|
|
253
|
+
required_suite_env(
|
|
254
|
+
child,
|
|
255
|
+
suite_path=_job_path(job, base_dir=base_dir),
|
|
256
|
+
)
|
|
257
|
+
)
|
|
258
|
+
continue
|
|
259
|
+
required.update(_as_string_list(child.get("required_env")))
|
|
260
|
+
return sorted(required)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def missing_suite_env(
|
|
264
|
+
suite: Mapping[str, Any],
|
|
265
|
+
*,
|
|
266
|
+
suite_path: str | Path = ".",
|
|
267
|
+
) -> list[str]:
|
|
268
|
+
return [
|
|
269
|
+
key
|
|
270
|
+
for key in required_suite_env(suite, suite_path=suite_path)
|
|
271
|
+
if not os.environ.get(key)
|
|
272
|
+
]
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def validate_suite_env(
|
|
276
|
+
suite: Mapping[str, Any],
|
|
277
|
+
*,
|
|
278
|
+
suite_path: str | Path = ".",
|
|
279
|
+
) -> None:
|
|
280
|
+
missing = missing_suite_env(suite, suite_path=suite_path)
|
|
281
|
+
if missing:
|
|
282
|
+
raise SuiteError(
|
|
283
|
+
"missing required environment variable(s): "
|
|
284
|
+
f"{', '.join(sorted(missing))}"
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def build_suite_manifest(
|
|
289
|
+
*,
|
|
290
|
+
name: str,
|
|
291
|
+
jobs: Sequence[Mapping[str, Any]],
|
|
292
|
+
required_env: Sequence[str] = (),
|
|
293
|
+
required_capabilities: Optional[Mapping[str, Sequence[str]]] = None,
|
|
294
|
+
outputs: Optional[Mapping[str, Any]] = None,
|
|
295
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
296
|
+
optimizer_governance_policy: Optional[Mapping[str, Any]] = None,
|
|
297
|
+
threshold: Optional[float] = None,
|
|
298
|
+
fail_fast: Optional[bool] = None,
|
|
299
|
+
) -> dict[str, Any]:
|
|
300
|
+
"""Build an Agent Learning suite manifest from SDK data.
|
|
301
|
+
|
|
302
|
+
This is the SDK counterpart to writing ``agent-learning.suite.v1`` JSON by
|
|
303
|
+
hand: users can compose run/eval/red-team/optimization jobs in Python and
|
|
304
|
+
execute them through ``run_suite`` or ``run_suite_file``.
|
|
305
|
+
"""
|
|
306
|
+
|
|
307
|
+
if not name:
|
|
308
|
+
raise ValueError("name is required")
|
|
309
|
+
if not jobs:
|
|
310
|
+
raise ValueError("jobs must contain at least one suite job")
|
|
311
|
+
manifest: dict[str, Any] = {
|
|
312
|
+
"version": AGENT_LEARNING_SUITE_KIND,
|
|
313
|
+
"name": str(name),
|
|
314
|
+
"required_env": _unique_strings(required_env),
|
|
315
|
+
"jobs": [
|
|
316
|
+
_normalize_suite_job(job, index)
|
|
317
|
+
for index, job in enumerate(jobs, start=1)
|
|
318
|
+
],
|
|
319
|
+
}
|
|
320
|
+
if required_capabilities:
|
|
321
|
+
manifest["required_capabilities"] = {
|
|
322
|
+
str(key): _unique_strings(value)
|
|
323
|
+
for key, value in dict(required_capabilities).items()
|
|
324
|
+
if _unique_strings(value)
|
|
325
|
+
}
|
|
326
|
+
if outputs:
|
|
327
|
+
manifest["outputs"] = copy.deepcopy(dict(outputs))
|
|
328
|
+
if metadata:
|
|
329
|
+
manifest["metadata"] = copy.deepcopy(dict(metadata))
|
|
330
|
+
if optimizer_governance_policy:
|
|
331
|
+
manifest["optimizer_governance_policy"] = copy.deepcopy(
|
|
332
|
+
dict(optimizer_governance_policy)
|
|
333
|
+
)
|
|
334
|
+
if threshold is not None:
|
|
335
|
+
manifest["threshold"] = float(threshold)
|
|
336
|
+
if fail_fast is not None:
|
|
337
|
+
manifest["fail_fast"] = bool(fail_fast)
|
|
338
|
+
return manifest
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def build_trinity_suite_manifest(
|
|
342
|
+
*,
|
|
343
|
+
name: str,
|
|
344
|
+
run_path: str | Path,
|
|
345
|
+
eval_path: str | Path,
|
|
346
|
+
artifact_eval_path: str | Path,
|
|
347
|
+
artifact_report_path: str | Path,
|
|
348
|
+
redteam_path: str | Path,
|
|
349
|
+
eval_optimization_path: str | Path,
|
|
350
|
+
optimization_path: str | Path,
|
|
351
|
+
world_model_optimization_path: str | Path | None = None,
|
|
352
|
+
artifact_action_id: str | None = "report_orchestration_strategy",
|
|
353
|
+
artifact_action_cwd: str | Path | None = "artifacts/action-loop/workspace",
|
|
354
|
+
artifact_optimization_path: str | Path | None = None,
|
|
355
|
+
artifact_eval_config_path: str | Path | None = None,
|
|
356
|
+
required_env: Sequence[str] = (),
|
|
357
|
+
max_candidates: Optional[int] = None,
|
|
358
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
359
|
+
) -> dict[str, Any]:
|
|
360
|
+
"""Build a run/eval/artifact/red-team/optimization suite.
|
|
361
|
+
|
|
362
|
+
The manifest mirrors the promptfoo-style trinity workflow: simulation,
|
|
363
|
+
text eval, saved-artifact eval, direct artifact-report eval, optional
|
|
364
|
+
artifact-evidence optimization, red-team, eval-suite optimization, and
|
|
365
|
+
full manifest optimization in one capability-gated suite.
|
|
366
|
+
"""
|
|
367
|
+
|
|
368
|
+
suite_name = str(name)
|
|
369
|
+
jobs: list[dict[str, Any]] = [
|
|
370
|
+
{
|
|
371
|
+
"id": "local-simulation",
|
|
372
|
+
"command": "run",
|
|
373
|
+
"path": _suite_path_text(run_path),
|
|
374
|
+
"name": f"{suite_name}-run",
|
|
375
|
+
},
|
|
376
|
+
{
|
|
377
|
+
"id": "promptfoo-style-eval",
|
|
378
|
+
"command": "eval",
|
|
379
|
+
"path": _suite_path_text(eval_path),
|
|
380
|
+
"name": f"{suite_name}-eval",
|
|
381
|
+
},
|
|
382
|
+
{
|
|
383
|
+
"id": "artifact-task-eval",
|
|
384
|
+
"command": "eval",
|
|
385
|
+
"path": _suite_path_text(artifact_eval_path),
|
|
386
|
+
"name": f"{suite_name}-artifact-eval",
|
|
387
|
+
},
|
|
388
|
+
{
|
|
389
|
+
"id": "direct-artifact-report-eval",
|
|
390
|
+
"command": "eval-artifact",
|
|
391
|
+
"path": _suite_path_text(artifact_report_path),
|
|
392
|
+
"name": f"{suite_name}-direct-artifact",
|
|
393
|
+
},
|
|
394
|
+
]
|
|
395
|
+
if artifact_action_id:
|
|
396
|
+
action_job = {
|
|
397
|
+
"id": "artifact-action-report",
|
|
398
|
+
"command": "action-run",
|
|
399
|
+
"path": _suite_path_text(artifact_report_path),
|
|
400
|
+
"action_id": str(artifact_action_id),
|
|
401
|
+
"name": f"{suite_name}-artifact-action-report",
|
|
402
|
+
"output": "../../artifacts/action-loop/action-run.json",
|
|
403
|
+
"outputs": {
|
|
404
|
+
"junit": "../../artifacts/action-loop/action-run.junit.xml",
|
|
405
|
+
"sarif": "../../artifacts/action-loop/action-run.sarif.json",
|
|
406
|
+
"markdown": "../../artifacts/action-loop/action-run.md",
|
|
407
|
+
},
|
|
408
|
+
}
|
|
409
|
+
if artifact_action_cwd is not None:
|
|
410
|
+
action_job["cwd"] = _suite_path_text(artifact_action_cwd)
|
|
411
|
+
jobs.append(action_job)
|
|
412
|
+
if artifact_optimization_path is not None:
|
|
413
|
+
jobs.append(
|
|
414
|
+
{
|
|
415
|
+
"id": "artifact-evidence-optimizer",
|
|
416
|
+
"command": "optimize-eval",
|
|
417
|
+
"path": _suite_path_text(artifact_optimization_path),
|
|
418
|
+
"name": f"{suite_name}-artifact-optimizer",
|
|
419
|
+
}
|
|
420
|
+
)
|
|
421
|
+
jobs.extend(
|
|
422
|
+
[
|
|
423
|
+
{
|
|
424
|
+
"id": "agent-red-team",
|
|
425
|
+
"command": "redteam",
|
|
426
|
+
"path": _suite_path_text(redteam_path),
|
|
427
|
+
"name": f"{suite_name}-redteam",
|
|
428
|
+
},
|
|
429
|
+
{
|
|
430
|
+
"id": "eval-suite-optimizer",
|
|
431
|
+
"command": "optimize-eval",
|
|
432
|
+
"path": _suite_path_text(eval_optimization_path),
|
|
433
|
+
"name": f"{suite_name}-eval-optimizer",
|
|
434
|
+
},
|
|
435
|
+
{
|
|
436
|
+
"id": "agent-optimizer",
|
|
437
|
+
"command": "optimize",
|
|
438
|
+
"path": _suite_path_text(optimization_path),
|
|
439
|
+
"name": f"{suite_name}-optimizer",
|
|
440
|
+
},
|
|
441
|
+
]
|
|
442
|
+
)
|
|
443
|
+
required_metrics = ["eval_assertions"]
|
|
444
|
+
if world_model_optimization_path is not None:
|
|
445
|
+
jobs.append(
|
|
446
|
+
{
|
|
447
|
+
"id": "world-model-optimizer",
|
|
448
|
+
"command": "optimize",
|
|
449
|
+
"path": _suite_path_text(world_model_optimization_path),
|
|
450
|
+
"name": f"{suite_name}-world-model-optimizer",
|
|
451
|
+
}
|
|
452
|
+
)
|
|
453
|
+
required_metrics.extend(
|
|
454
|
+
[
|
|
455
|
+
"world_contract_quality",
|
|
456
|
+
"world_contract_coverage",
|
|
457
|
+
"tool_selection_accuracy",
|
|
458
|
+
]
|
|
459
|
+
)
|
|
460
|
+
if artifact_eval_config_path is not None:
|
|
461
|
+
jobs[3]["config"] = _suite_path_text(artifact_eval_config_path)
|
|
462
|
+
if max_candidates is not None:
|
|
463
|
+
for job in jobs:
|
|
464
|
+
if job["command"] in {"optimize", "optimize-eval"}:
|
|
465
|
+
job["max_candidates"] = int(max_candidates)
|
|
466
|
+
return build_suite_manifest(
|
|
467
|
+
name=suite_name,
|
|
468
|
+
required_env=required_env,
|
|
469
|
+
jobs=jobs,
|
|
470
|
+
required_capabilities={
|
|
471
|
+
"commands": [
|
|
472
|
+
"run",
|
|
473
|
+
"eval",
|
|
474
|
+
"eval_artifact",
|
|
475
|
+
"action_run",
|
|
476
|
+
"redteam",
|
|
477
|
+
"optimize_eval",
|
|
478
|
+
"optimize",
|
|
479
|
+
],
|
|
480
|
+
"result_kinds": [
|
|
481
|
+
"agent-learning.run.v1",
|
|
482
|
+
"agent-learning.eval.v1",
|
|
483
|
+
"agent-learning.artifact-evaluation.v1",
|
|
484
|
+
"agent-learning.action-run.v1",
|
|
485
|
+
"agent-learning.redteam.v1",
|
|
486
|
+
"agent-learning.eval-optimization.v1",
|
|
487
|
+
"agent-learning.optimization.v1",
|
|
488
|
+
],
|
|
489
|
+
"metrics": required_metrics,
|
|
490
|
+
},
|
|
491
|
+
metadata={
|
|
492
|
+
"source": "fi.alk.suite.build_trinity_suite_manifest",
|
|
493
|
+
**copy.deepcopy(dict(metadata or {})),
|
|
494
|
+
},
|
|
495
|
+
optimizer_governance_policy={
|
|
496
|
+
"require_optimizer_governance": True,
|
|
497
|
+
"min_governed": 1,
|
|
498
|
+
},
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def build_framework_adapter_trinity_suite_manifest(
|
|
503
|
+
*,
|
|
504
|
+
name: str,
|
|
505
|
+
run_path: str | Path,
|
|
506
|
+
redteam_path: str | Path,
|
|
507
|
+
required_env: Sequence[str] = (),
|
|
508
|
+
required_frameworks: Sequence[str] = (),
|
|
509
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
510
|
+
outputs: Optional[Mapping[str, Any]] = None,
|
|
511
|
+
threshold: Optional[float] = None,
|
|
512
|
+
fail_fast: bool = True,
|
|
513
|
+
) -> dict[str, Any]:
|
|
514
|
+
"""Build a focused suite for framework simulation, eval, and red-team gates."""
|
|
515
|
+
|
|
516
|
+
if not name:
|
|
517
|
+
raise ValueError("name is required")
|
|
518
|
+
suite_name = str(name)
|
|
519
|
+
frameworks = _unique_strings(required_frameworks)
|
|
520
|
+
required_capabilities: dict[str, list[str]] = {
|
|
521
|
+
"commands": ["run", "redteam"],
|
|
522
|
+
"result_kinds": [
|
|
523
|
+
"agent-learning.run.v1",
|
|
524
|
+
"agent-learning.redteam.v1",
|
|
525
|
+
],
|
|
526
|
+
"metrics": [
|
|
527
|
+
"framework_runtime_contract",
|
|
528
|
+
"framework_adapter_contract_quality",
|
|
529
|
+
"adversarial_resilience",
|
|
530
|
+
"red_team_campaign_quality",
|
|
531
|
+
],
|
|
532
|
+
}
|
|
533
|
+
if frameworks:
|
|
534
|
+
required_capabilities["frameworks"] = frameworks
|
|
535
|
+
return build_suite_manifest(
|
|
536
|
+
name=suite_name,
|
|
537
|
+
required_env=required_env,
|
|
538
|
+
jobs=[
|
|
539
|
+
{
|
|
540
|
+
"id": "optimized-framework-run",
|
|
541
|
+
"command": "run",
|
|
542
|
+
"path": _suite_path_text(run_path),
|
|
543
|
+
"name": f"{suite_name}-run",
|
|
544
|
+
},
|
|
545
|
+
{
|
|
546
|
+
"id": "framework-red-team",
|
|
547
|
+
"command": "redteam",
|
|
548
|
+
"path": _suite_path_text(redteam_path),
|
|
549
|
+
"name": f"{suite_name}-redteam",
|
|
550
|
+
},
|
|
551
|
+
],
|
|
552
|
+
required_capabilities=required_capabilities,
|
|
553
|
+
outputs=outputs,
|
|
554
|
+
threshold=threshold,
|
|
555
|
+
fail_fast=fail_fast,
|
|
556
|
+
metadata={
|
|
557
|
+
"source": "fi.alk.suite.build_framework_adapter_trinity_suite_manifest",
|
|
558
|
+
"task_kind": "framework_adapter_trinity_suite",
|
|
559
|
+
**copy.deepcopy(dict(metadata or {})),
|
|
560
|
+
},
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def write_framework_adapter_trinity_suite_workspace(
|
|
565
|
+
*,
|
|
566
|
+
name: str,
|
|
567
|
+
framework: str,
|
|
568
|
+
target: str,
|
|
569
|
+
directory: str | Path,
|
|
570
|
+
adapter_candidates: Optional[Sequence[Mapping[str, Any]]] = None,
|
|
571
|
+
agent: Any = None,
|
|
572
|
+
agent_factory: Any = None,
|
|
573
|
+
cases: Sequence[Mapping[str, Any]] = (),
|
|
574
|
+
target_base_dir: str | Path = ".",
|
|
575
|
+
target_factory: Optional[bool] = None,
|
|
576
|
+
method_candidates: Optional[Sequence[str | None]] = None,
|
|
577
|
+
input_mode_candidates: Optional[Sequence[str]] = None,
|
|
578
|
+
required_env: Sequence[str] = (),
|
|
579
|
+
scenario: Optional[Mapping[str, Any]] = None,
|
|
580
|
+
framework_trace: Optional[Mapping[str, Any]] = None,
|
|
581
|
+
evaluation_config: Optional[Mapping[str, Any]] = None,
|
|
582
|
+
auto_evaluation_config: bool = True,
|
|
583
|
+
threshold: float = 0.9,
|
|
584
|
+
trace_runtime: bool = True,
|
|
585
|
+
allow_external_target: bool = False,
|
|
586
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
587
|
+
discovery_max_candidates: Optional[int] = 8,
|
|
588
|
+
max_candidates: Optional[int] = None,
|
|
589
|
+
include_seed: bool = True,
|
|
590
|
+
factory: Optional[bool] = None,
|
|
591
|
+
min_turns: int = 1,
|
|
592
|
+
max_turns: int = 1,
|
|
593
|
+
redteam_attacks: Sequence[str] = ("prompt_injection", "credential_exfiltration"),
|
|
594
|
+
redteam_surfaces: Sequence[str] = ("instruction", "tool"),
|
|
595
|
+
redteam_taxonomies: Sequence[str] = ("owasp_llm_top_10", "owasp_agentic_ai"),
|
|
596
|
+
redteam_channels: Sequence[str] = ("chat",),
|
|
597
|
+
redteam_providers: Sequence[str] = ("local_cli",),
|
|
598
|
+
redteam_agent: Optional[Mapping[str, Any]] = None,
|
|
599
|
+
redteam_config: Optional[Mapping[str, Any]] = None,
|
|
600
|
+
redteam_overrides: Optional[Mapping[str, Any]] = None,
|
|
601
|
+
canaries: Sequence[Any] = (),
|
|
602
|
+
blocked_tools: Sequence[str] = (),
|
|
603
|
+
redteam_min_turns: int = 3,
|
|
604
|
+
redteam_max_turns: int = 3,
|
|
605
|
+
) -> dict[str, Any]:
|
|
606
|
+
"""Write a runnable framework adapter run+red-team suite workspace."""
|
|
607
|
+
|
|
608
|
+
if not name:
|
|
609
|
+
raise ValueError("name is required")
|
|
610
|
+
if not framework:
|
|
611
|
+
raise ValueError("framework is required")
|
|
612
|
+
if not target:
|
|
613
|
+
raise ValueError("target is required")
|
|
614
|
+
|
|
615
|
+
workspace = Path(directory).expanduser().resolve()
|
|
616
|
+
manifests_dir = workspace / "manifests"
|
|
617
|
+
manifests_dir.mkdir(parents=True, exist_ok=True)
|
|
618
|
+
selected_target = _suite_local_target_text(target, base_dir=target_base_dir)
|
|
619
|
+
suite_metadata = copy.deepcopy(dict(metadata or {}))
|
|
620
|
+
|
|
621
|
+
from fi.alk import optimize, redteam
|
|
622
|
+
|
|
623
|
+
run_manifest = optimize.build_framework_run_manifest_from_local_adapter(
|
|
624
|
+
name=f"{name}-run",
|
|
625
|
+
framework=framework,
|
|
626
|
+
target=selected_target,
|
|
627
|
+
adapter_candidates=adapter_candidates,
|
|
628
|
+
agent=agent,
|
|
629
|
+
agent_factory=agent_factory,
|
|
630
|
+
cases=cases,
|
|
631
|
+
target_base_dir=target_base_dir,
|
|
632
|
+
target_factory=target_factory,
|
|
633
|
+
method_candidates=method_candidates,
|
|
634
|
+
input_mode_candidates=input_mode_candidates,
|
|
635
|
+
required_env=required_env,
|
|
636
|
+
scenario=scenario,
|
|
637
|
+
framework_trace=framework_trace,
|
|
638
|
+
evaluation_config=evaluation_config,
|
|
639
|
+
auto_evaluation_config=auto_evaluation_config,
|
|
640
|
+
threshold=threshold,
|
|
641
|
+
trace_runtime=trace_runtime,
|
|
642
|
+
allow_external_target=allow_external_target,
|
|
643
|
+
metadata={
|
|
644
|
+
"suite": name,
|
|
645
|
+
"suite_role": "optimized_framework_run",
|
|
646
|
+
**suite_metadata,
|
|
647
|
+
},
|
|
648
|
+
discovery_max_candidates=discovery_max_candidates,
|
|
649
|
+
max_candidates=max_candidates,
|
|
650
|
+
include_seed=include_seed,
|
|
651
|
+
factory=factory,
|
|
652
|
+
min_turns=min_turns,
|
|
653
|
+
max_turns=max_turns,
|
|
654
|
+
)
|
|
655
|
+
run_path = _write_suite_json(
|
|
656
|
+
run_manifest,
|
|
657
|
+
manifests_dir / "optimized-framework-run.json",
|
|
658
|
+
)
|
|
659
|
+
|
|
660
|
+
agent_config = copy.deepcopy(dict(run_manifest.get("agent") or {}))
|
|
661
|
+
agent_metadata = copy.deepcopy(dict(agent_config.get("metadata") or {}))
|
|
662
|
+
redteam_manifest = redteam.build_redteam_manifest(
|
|
663
|
+
name=f"{name}-redteam",
|
|
664
|
+
attacks=redteam_attacks,
|
|
665
|
+
surfaces=redteam_surfaces,
|
|
666
|
+
taxonomies=redteam_taxonomies,
|
|
667
|
+
channels=redteam_channels,
|
|
668
|
+
providers=redteam_providers,
|
|
669
|
+
frameworks=[framework],
|
|
670
|
+
required_env=required_env,
|
|
671
|
+
target={
|
|
672
|
+
"agent": run_manifest.get("name") or f"{name}-run",
|
|
673
|
+
"framework": framework,
|
|
674
|
+
"adapter_target": selected_target,
|
|
675
|
+
"framework_adapter_contract": copy.deepcopy(
|
|
676
|
+
agent_metadata.get("framework_adapter_probe_contract")
|
|
677
|
+
or agent_metadata.get("framework_adapter_contract")
|
|
678
|
+
),
|
|
679
|
+
"framework_adapter_probe_proof_status": (
|
|
680
|
+
copy.deepcopy(
|
|
681
|
+
dict(agent_metadata.get("framework_adapter_probe_proof") or {})
|
|
682
|
+
).get("status")
|
|
683
|
+
),
|
|
684
|
+
"framework_adapter_discovery_used": bool(
|
|
685
|
+
agent_metadata.get("framework_adapter_discovery_used")
|
|
686
|
+
),
|
|
687
|
+
"suite": name,
|
|
688
|
+
},
|
|
689
|
+
agent=redteam_agent,
|
|
690
|
+
redteam=redteam_overrides,
|
|
691
|
+
evaluation_config=redteam_config,
|
|
692
|
+
threshold=threshold,
|
|
693
|
+
canaries=canaries,
|
|
694
|
+
blocked_tools=blocked_tools,
|
|
695
|
+
min_turns=redteam_min_turns,
|
|
696
|
+
max_turns=redteam_max_turns,
|
|
697
|
+
)
|
|
698
|
+
redteam_path = _write_suite_json(
|
|
699
|
+
redteam_manifest,
|
|
700
|
+
manifests_dir / "framework-redteam.json",
|
|
701
|
+
)
|
|
702
|
+
|
|
703
|
+
suite_manifest = build_framework_adapter_trinity_suite_manifest(
|
|
704
|
+
name=name,
|
|
705
|
+
run_path=Path("manifests") / run_path.name,
|
|
706
|
+
redteam_path=Path("manifests") / redteam_path.name,
|
|
707
|
+
required_env=required_env,
|
|
708
|
+
required_frameworks=[framework],
|
|
709
|
+
threshold=threshold,
|
|
710
|
+
metadata={
|
|
711
|
+
"source": "fi.alk.suite.write_framework_adapter_trinity_suite_workspace",
|
|
712
|
+
"framework": framework,
|
|
713
|
+
"target": selected_target,
|
|
714
|
+
**suite_metadata,
|
|
715
|
+
},
|
|
716
|
+
)
|
|
717
|
+
suite_path = write_suite_file(suite_manifest, workspace / "suite.json")
|
|
718
|
+
return {
|
|
719
|
+
"kind": "agent-learning.framework-adapter-trinity-workspace.v1",
|
|
720
|
+
"status": "passed",
|
|
721
|
+
"name": str(name),
|
|
722
|
+
"summary": {
|
|
723
|
+
"framework": framework,
|
|
724
|
+
"target": selected_target,
|
|
725
|
+
"suite_job_count": len(suite_manifest["jobs"]),
|
|
726
|
+
"run_manifest": str(run_path),
|
|
727
|
+
"redteam_manifest": str(redteam_path),
|
|
728
|
+
"suite_manifest": str(suite_path),
|
|
729
|
+
},
|
|
730
|
+
"paths": {
|
|
731
|
+
"workspace": str(workspace),
|
|
732
|
+
"suite": str(suite_path),
|
|
733
|
+
"run": str(run_path),
|
|
734
|
+
"redteam": str(redteam_path),
|
|
735
|
+
},
|
|
736
|
+
"suite": suite_manifest,
|
|
737
|
+
"run_manifest": run_manifest,
|
|
738
|
+
"redteam_manifest": redteam_manifest,
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def build_framework_adapter_trinity_suite_optimization_manifest(
|
|
743
|
+
*,
|
|
744
|
+
name: str,
|
|
745
|
+
run_path: str | Path,
|
|
746
|
+
trinity_suite_path: str | Path,
|
|
747
|
+
framework: str,
|
|
748
|
+
required_env: Sequence[str] = (),
|
|
749
|
+
required_frameworks: Sequence[str] = (),
|
|
750
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
751
|
+
threshold: float = 1.0,
|
|
752
|
+
optimizer: Optional[Mapping[str, Any]] = None,
|
|
753
|
+
) -> dict[str, Any]:
|
|
754
|
+
"""Build a suite optimization that selects full framework trinity coverage."""
|
|
755
|
+
|
|
756
|
+
if not name:
|
|
757
|
+
raise ValueError("name is required")
|
|
758
|
+
if not framework:
|
|
759
|
+
raise ValueError("framework is required")
|
|
760
|
+
suite_name = str(name)
|
|
761
|
+
frameworks = _unique_strings(required_frameworks) or [str(framework)]
|
|
762
|
+
seed_job = {
|
|
763
|
+
"id": "optimized-framework-run",
|
|
764
|
+
"command": "run",
|
|
765
|
+
"path": _suite_path_text(run_path),
|
|
766
|
+
"name": f"{suite_name}-run-only-seed",
|
|
767
|
+
}
|
|
768
|
+
trinity_job = {
|
|
769
|
+
"id": "framework-adapter-trinity",
|
|
770
|
+
"command": "suite",
|
|
771
|
+
"path": _suite_path_text(trinity_suite_path),
|
|
772
|
+
"name": f"{suite_name}-full-trinity",
|
|
773
|
+
}
|
|
774
|
+
manifest = build_suite_manifest(
|
|
775
|
+
name=suite_name,
|
|
776
|
+
required_env=required_env,
|
|
777
|
+
jobs=[seed_job],
|
|
778
|
+
required_capabilities={
|
|
779
|
+
"commands": ["run", "redteam", "suite"],
|
|
780
|
+
"result_kinds": [
|
|
781
|
+
"agent-learning.run.v1",
|
|
782
|
+
"agent-learning.redteam.v1",
|
|
783
|
+
"agent-learning.suite.v1",
|
|
784
|
+
],
|
|
785
|
+
"frameworks": frameworks,
|
|
786
|
+
"metrics": [
|
|
787
|
+
"framework_runtime_contract",
|
|
788
|
+
"framework_adapter_contract_quality",
|
|
789
|
+
"adversarial_resilience",
|
|
790
|
+
"red_team_campaign_quality",
|
|
791
|
+
],
|
|
792
|
+
},
|
|
793
|
+
metadata={
|
|
794
|
+
"source": (
|
|
795
|
+
"fi.alk.suite."
|
|
796
|
+
"build_framework_adapter_trinity_suite_optimization_manifest"
|
|
797
|
+
),
|
|
798
|
+
"task_kind": "framework_adapter_trinity_suite_optimization",
|
|
799
|
+
"framework": framework,
|
|
800
|
+
**copy.deepcopy(dict(metadata or {})),
|
|
801
|
+
},
|
|
802
|
+
)
|
|
803
|
+
manifest["optimization"] = {
|
|
804
|
+
"threshold": float(threshold),
|
|
805
|
+
"target": {
|
|
806
|
+
"name": suite_name,
|
|
807
|
+
"layers": ["harness", "framework", "security", "evaluator"],
|
|
808
|
+
"base_config": {"jobs": [copy.deepcopy(seed_job)]},
|
|
809
|
+
"search_space": {
|
|
810
|
+
"jobs.0": [
|
|
811
|
+
copy.deepcopy(seed_job),
|
|
812
|
+
copy.deepcopy(trinity_job),
|
|
813
|
+
]
|
|
814
|
+
},
|
|
815
|
+
"metadata": {
|
|
816
|
+
"source": (
|
|
817
|
+
"fi.alk.suite."
|
|
818
|
+
"build_framework_adapter_trinity_suite_optimization_manifest"
|
|
819
|
+
),
|
|
820
|
+
"task_kind": "framework_adapter_trinity_suite_optimization",
|
|
821
|
+
"framework": framework,
|
|
822
|
+
**copy.deepcopy(dict(metadata or {})),
|
|
823
|
+
},
|
|
824
|
+
},
|
|
825
|
+
"optimizer": copy.deepcopy(
|
|
826
|
+
dict(
|
|
827
|
+
optimizer
|
|
828
|
+
or {
|
|
829
|
+
"algorithm": "agent",
|
|
830
|
+
"max_candidates": 3,
|
|
831
|
+
"include_seed": True,
|
|
832
|
+
"auto_diagnose": False,
|
|
833
|
+
}
|
|
834
|
+
)
|
|
835
|
+
),
|
|
836
|
+
}
|
|
837
|
+
return manifest
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def write_framework_adapter_trinity_suite_optimization_workspace(
|
|
841
|
+
*,
|
|
842
|
+
name: str,
|
|
843
|
+
framework: str,
|
|
844
|
+
target: str,
|
|
845
|
+
directory: str | Path,
|
|
846
|
+
suite_optimization_threshold: float = 1.0,
|
|
847
|
+
suite_optimizer: Optional[Mapping[str, Any]] = None,
|
|
848
|
+
**workspace_kwargs: Any,
|
|
849
|
+
) -> dict[str, Any]:
|
|
850
|
+
"""Write a framework trinity workspace plus an optimizable outer suite."""
|
|
851
|
+
|
|
852
|
+
workspace = write_framework_adapter_trinity_suite_workspace(
|
|
853
|
+
name=name,
|
|
854
|
+
framework=framework,
|
|
855
|
+
target=target,
|
|
856
|
+
directory=directory,
|
|
857
|
+
**workspace_kwargs,
|
|
858
|
+
)
|
|
859
|
+
workspace_root = Path(workspace["paths"]["workspace"]).expanduser().resolve()
|
|
860
|
+
metadata = copy.deepcopy(dict(workspace_kwargs.get("metadata") or {}))
|
|
861
|
+
optimization_manifest = build_framework_adapter_trinity_suite_optimization_manifest(
|
|
862
|
+
name=f"{name}-optimization",
|
|
863
|
+
run_path=Path("manifests") / Path(workspace["paths"]["run"]).name,
|
|
864
|
+
trinity_suite_path=Path("suite.json"),
|
|
865
|
+
framework=framework,
|
|
866
|
+
required_env=workspace_kwargs.get("required_env", ()),
|
|
867
|
+
required_frameworks=[framework],
|
|
868
|
+
metadata={
|
|
869
|
+
"source": (
|
|
870
|
+
"fi.alk.suite."
|
|
871
|
+
"write_framework_adapter_trinity_suite_optimization_workspace"
|
|
872
|
+
),
|
|
873
|
+
"framework": framework,
|
|
874
|
+
"target": workspace["summary"]["target"],
|
|
875
|
+
**metadata,
|
|
876
|
+
},
|
|
877
|
+
threshold=suite_optimization_threshold,
|
|
878
|
+
optimizer=suite_optimizer,
|
|
879
|
+
)
|
|
880
|
+
optimization_path = write_suite_file(
|
|
881
|
+
optimization_manifest,
|
|
882
|
+
workspace_root / "suite-optimization.json",
|
|
883
|
+
)
|
|
884
|
+
return {
|
|
885
|
+
"kind": "agent-learning.framework-adapter-trinity-optimization-workspace.v1",
|
|
886
|
+
"status": "passed",
|
|
887
|
+
"name": str(name),
|
|
888
|
+
"summary": {
|
|
889
|
+
**copy.deepcopy(dict(workspace.get("summary") or {})),
|
|
890
|
+
"suite_optimization_manifest": str(optimization_path),
|
|
891
|
+
"suite_optimization_search_paths": ["jobs.0"],
|
|
892
|
+
},
|
|
893
|
+
"paths": {
|
|
894
|
+
**copy.deepcopy(dict(workspace.get("paths") or {})),
|
|
895
|
+
"suite_optimization": str(optimization_path),
|
|
896
|
+
},
|
|
897
|
+
"suite_optimization": optimization_manifest,
|
|
898
|
+
"trinity_workspace": workspace,
|
|
899
|
+
}
|
|
900
|
+
|
|
901
|
+
|
|
902
|
+
def build_regression_artifact_suite_manifest(
|
|
903
|
+
*,
|
|
904
|
+
name: str,
|
|
905
|
+
baseline_path: str | Path,
|
|
906
|
+
current_path: str | Path,
|
|
907
|
+
finding_path: str | Path,
|
|
908
|
+
replay_manifest_paths: Sequence[str | Path],
|
|
909
|
+
required_env: Sequence[str] = (),
|
|
910
|
+
min_score_delta: float = 0.0,
|
|
911
|
+
max_new_findings: int = 0,
|
|
912
|
+
max_new_error_findings: int = 0,
|
|
913
|
+
min_level: str = "warning",
|
|
914
|
+
max_findings: int = 1,
|
|
915
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
916
|
+
) -> dict[str, Any]:
|
|
917
|
+
"""Build the artifact-regression lifecycle suite from SDK paths.
|
|
918
|
+
|
|
919
|
+
This composes the lifecycle users usually script around CI artifacts:
|
|
920
|
+
create a compact baseline, compare current vs baseline, render a report,
|
|
921
|
+
promote a red-team finding into a regression manifest, and replay one or
|
|
922
|
+
more regression manifests.
|
|
923
|
+
"""
|
|
924
|
+
|
|
925
|
+
replay_paths = [_suite_path_text(path) for path in replay_manifest_paths]
|
|
926
|
+
if not replay_paths:
|
|
927
|
+
raise ValueError("replay_manifest_paths must contain at least one manifest")
|
|
928
|
+
|
|
929
|
+
suite_name = str(name)
|
|
930
|
+
jobs = [
|
|
931
|
+
{
|
|
932
|
+
"id": "baseline-current-run",
|
|
933
|
+
"command": "baseline",
|
|
934
|
+
"path": _suite_path_text(current_path),
|
|
935
|
+
"name": f"{suite_name}-baseline",
|
|
936
|
+
},
|
|
937
|
+
{
|
|
938
|
+
"id": "compare-baseline-to-current",
|
|
939
|
+
"command": "compare",
|
|
940
|
+
"path": _suite_path_text(current_path),
|
|
941
|
+
"baseline": _suite_path_text(baseline_path),
|
|
942
|
+
"current": _suite_path_text(current_path),
|
|
943
|
+
"name": f"{suite_name}-compare",
|
|
944
|
+
"min_score_delta": float(min_score_delta),
|
|
945
|
+
"max_new_findings": int(max_new_findings),
|
|
946
|
+
"max_new_error_findings": int(max_new_error_findings),
|
|
947
|
+
},
|
|
948
|
+
{
|
|
949
|
+
"id": "report-current-run",
|
|
950
|
+
"command": "report",
|
|
951
|
+
"path": _suite_path_text(current_path),
|
|
952
|
+
"name": f"{suite_name}-report",
|
|
953
|
+
},
|
|
954
|
+
{
|
|
955
|
+
"id": "promote-redteam-finding",
|
|
956
|
+
"command": "promote_to_regression",
|
|
957
|
+
"path": _suite_path_text(finding_path),
|
|
958
|
+
"name": f"{suite_name}-promoted-regression",
|
|
959
|
+
"min_level": str(min_level),
|
|
960
|
+
"max_findings": int(max_findings),
|
|
961
|
+
},
|
|
962
|
+
{
|
|
963
|
+
"id": "replay-regression-manifest",
|
|
964
|
+
"command": "replay",
|
|
965
|
+
"path": replay_paths[0],
|
|
966
|
+
"manifests": replay_paths,
|
|
967
|
+
"name": f"{suite_name}-replay",
|
|
968
|
+
},
|
|
969
|
+
]
|
|
970
|
+
return build_suite_manifest(
|
|
971
|
+
name=suite_name,
|
|
972
|
+
required_env=required_env,
|
|
973
|
+
jobs=jobs,
|
|
974
|
+
required_capabilities={
|
|
975
|
+
"commands": [
|
|
976
|
+
"baseline",
|
|
977
|
+
"compare",
|
|
978
|
+
"report",
|
|
979
|
+
"promote_to_regression",
|
|
980
|
+
"replay",
|
|
981
|
+
],
|
|
982
|
+
"result_kinds": [
|
|
983
|
+
"agent_learning.baseline.v1",
|
|
984
|
+
"agent_learning.compare.v1",
|
|
985
|
+
"agent_learning.report.v1",
|
|
986
|
+
"agent_learning.regression_promotion.v1",
|
|
987
|
+
"agent_learning.replay.v1",
|
|
988
|
+
],
|
|
989
|
+
"metrics": [
|
|
990
|
+
"compare_score_delta",
|
|
991
|
+
"replay_pass_rate",
|
|
992
|
+
],
|
|
993
|
+
},
|
|
994
|
+
metadata={
|
|
995
|
+
"source": "fi.alk.suite.build_regression_artifact_suite_manifest",
|
|
996
|
+
"task_kind": "regression_artifact_lifecycle",
|
|
997
|
+
**copy.deepcopy(dict(metadata or {})),
|
|
998
|
+
},
|
|
999
|
+
)
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def build_optimization_lifecycle_plan(
|
|
1003
|
+
*,
|
|
1004
|
+
optimize_manifest_path: str | Path,
|
|
1005
|
+
workspace_dir: str | Path | None = None,
|
|
1006
|
+
name: str = "optimization-lifecycle",
|
|
1007
|
+
required_env: Sequence[str] = (),
|
|
1008
|
+
frozen_profile_path: str | Path | None = None,
|
|
1009
|
+
) -> dict[str, Any]:
|
|
1010
|
+
"""Build an executable optimize -> promote -> replay lifecycle plan.
|
|
1011
|
+
|
|
1012
|
+
When ``frozen_profile_path`` names a frozen capability-profile contract
|
|
1013
|
+
(kind ``agent-learning.frozen-capability-profile.v1``, ARCH §2a), the plan
|
|
1014
|
+
gains a ``replay_frozen_profile`` step between the promotion and the
|
|
1015
|
+
regression replay: every frozen row is re-closed against the optimization
|
|
1016
|
+
artifact and an improving-but-row-breaking candidate is vetoed
|
|
1017
|
+
(hetvabhasa class ``badhita``) before any replay runs.
|
|
1018
|
+
"""
|
|
1019
|
+
|
|
1020
|
+
paths = _optimization_lifecycle_paths(
|
|
1021
|
+
optimize_manifest_path=optimize_manifest_path,
|
|
1022
|
+
workspace_dir=workspace_dir,
|
|
1023
|
+
)
|
|
1024
|
+
if frozen_profile_path is not None:
|
|
1025
|
+
frozen_path = Path(frozen_profile_path).expanduser().resolve()
|
|
1026
|
+
paths["frozen_profile"] = frozen_path
|
|
1027
|
+
paths["frozen_profile_replay"] = (
|
|
1028
|
+
paths["optimization"].parent / "frozen-profile-replay.json"
|
|
1029
|
+
)
|
|
1030
|
+
required_env_args = _required_env_cli_args(required_env)
|
|
1031
|
+
steps = [
|
|
1032
|
+
_lifecycle_step(
|
|
1033
|
+
"dry_run_optimization",
|
|
1034
|
+
"Dry Run Optimization",
|
|
1035
|
+
["agent-learn", "optimize", paths["optimize_manifest"], "--dry-run"],
|
|
1036
|
+
),
|
|
1037
|
+
_lifecycle_step(
|
|
1038
|
+
"optimize",
|
|
1039
|
+
"Run Optimization",
|
|
1040
|
+
[
|
|
1041
|
+
"agent-learn",
|
|
1042
|
+
"optimize",
|
|
1043
|
+
paths["optimize_manifest"],
|
|
1044
|
+
"--output",
|
|
1045
|
+
paths["optimization"],
|
|
1046
|
+
"--junit",
|
|
1047
|
+
paths["optimization_junit"],
|
|
1048
|
+
"--sarif",
|
|
1049
|
+
paths["optimization_sarif"],
|
|
1050
|
+
"--markdown",
|
|
1051
|
+
paths["optimization_markdown"],
|
|
1052
|
+
],
|
|
1053
|
+
outputs={
|
|
1054
|
+
"json": paths["optimization"],
|
|
1055
|
+
"junit": paths["optimization_junit"],
|
|
1056
|
+
"sarif": paths["optimization_sarif"],
|
|
1057
|
+
"markdown": paths["optimization_markdown"],
|
|
1058
|
+
},
|
|
1059
|
+
),
|
|
1060
|
+
_lifecycle_step(
|
|
1061
|
+
"report_optimization",
|
|
1062
|
+
"Report Optimization",
|
|
1063
|
+
[
|
|
1064
|
+
"agent-learn",
|
|
1065
|
+
"report",
|
|
1066
|
+
paths["optimization"],
|
|
1067
|
+
"--output",
|
|
1068
|
+
paths["optimization_report"],
|
|
1069
|
+
"--markdown",
|
|
1070
|
+
paths["optimization_report_markdown"],
|
|
1071
|
+
],
|
|
1072
|
+
outputs={
|
|
1073
|
+
"json": paths["optimization_report"],
|
|
1074
|
+
"markdown": paths["optimization_report_markdown"],
|
|
1075
|
+
},
|
|
1076
|
+
),
|
|
1077
|
+
_lifecycle_step(
|
|
1078
|
+
"promote_to_regression",
|
|
1079
|
+
"Promote To Regression",
|
|
1080
|
+
[
|
|
1081
|
+
"agent-learn",
|
|
1082
|
+
"promote-to-regression",
|
|
1083
|
+
paths["optimization"],
|
|
1084
|
+
"--output",
|
|
1085
|
+
paths["promotion"],
|
|
1086
|
+
"--manifest",
|
|
1087
|
+
paths["regression_manifest"],
|
|
1088
|
+
"--min-level",
|
|
1089
|
+
"note",
|
|
1090
|
+
"--max-findings",
|
|
1091
|
+
"1",
|
|
1092
|
+
*required_env_args,
|
|
1093
|
+
],
|
|
1094
|
+
outputs={
|
|
1095
|
+
"json": paths["promotion"],
|
|
1096
|
+
"manifest": paths["regression_manifest"],
|
|
1097
|
+
},
|
|
1098
|
+
),
|
|
1099
|
+
_lifecycle_step(
|
|
1100
|
+
"report_promotion",
|
|
1101
|
+
"Report Promotion",
|
|
1102
|
+
[
|
|
1103
|
+
"agent-learn",
|
|
1104
|
+
"report",
|
|
1105
|
+
paths["promotion"],
|
|
1106
|
+
"--output",
|
|
1107
|
+
paths["promotion_report"],
|
|
1108
|
+
"--markdown",
|
|
1109
|
+
paths["promotion_report_markdown"],
|
|
1110
|
+
],
|
|
1111
|
+
outputs={
|
|
1112
|
+
"json": paths["promotion_report"],
|
|
1113
|
+
"markdown": paths["promotion_report_markdown"],
|
|
1114
|
+
},
|
|
1115
|
+
),
|
|
1116
|
+
*(
|
|
1117
|
+
[
|
|
1118
|
+
_lifecycle_step(
|
|
1119
|
+
"replay_frozen_profile",
|
|
1120
|
+
"Replay Frozen Capability Profile",
|
|
1121
|
+
[
|
|
1122
|
+
sys.executable,
|
|
1123
|
+
"-c",
|
|
1124
|
+
(
|
|
1125
|
+
"import json, pathlib; "
|
|
1126
|
+
"from fi.alk import optimize; "
|
|
1127
|
+
"result = json.loads(pathlib.Path("
|
|
1128
|
+
f"{str(paths['optimization'])!r}"
|
|
1129
|
+
").read_text(encoding='utf-8')); "
|
|
1130
|
+
"frozen = json.loads(pathlib.Path("
|
|
1131
|
+
f"{str(paths['frozen_profile'])!r}"
|
|
1132
|
+
").read_text(encoding='utf-8')); "
|
|
1133
|
+
"verdict = optimize.replay_frozen_profile(result, frozen); "
|
|
1134
|
+
"pathlib.Path("
|
|
1135
|
+
f"{str(paths['frozen_profile_replay'])!r}"
|
|
1136
|
+
").write_text(json.dumps(verdict, indent=2, "
|
|
1137
|
+
"sort_keys=True, default=str), encoding='utf-8'); "
|
|
1138
|
+
"raise SystemExit(1 if verdict.get('veto') else 0)"
|
|
1139
|
+
),
|
|
1140
|
+
],
|
|
1141
|
+
outputs={"json": paths["frozen_profile_replay"]},
|
|
1142
|
+
)
|
|
1143
|
+
]
|
|
1144
|
+
if frozen_profile_path is not None
|
|
1145
|
+
else []
|
|
1146
|
+
),
|
|
1147
|
+
_lifecycle_step(
|
|
1148
|
+
"replay_regression",
|
|
1149
|
+
"Replay Regression",
|
|
1150
|
+
[
|
|
1151
|
+
"agent-learn",
|
|
1152
|
+
"replay",
|
|
1153
|
+
paths["regression_manifest"],
|
|
1154
|
+
"--output",
|
|
1155
|
+
paths["replay"],
|
|
1156
|
+
"--junit",
|
|
1157
|
+
paths["replay_junit"],
|
|
1158
|
+
"--sarif",
|
|
1159
|
+
paths["replay_sarif"],
|
|
1160
|
+
"--markdown",
|
|
1161
|
+
paths["replay_markdown"],
|
|
1162
|
+
],
|
|
1163
|
+
outputs={
|
|
1164
|
+
"json": paths["replay"],
|
|
1165
|
+
"junit": paths["replay_junit"],
|
|
1166
|
+
"sarif": paths["replay_sarif"],
|
|
1167
|
+
"markdown": paths["replay_markdown"],
|
|
1168
|
+
},
|
|
1169
|
+
),
|
|
1170
|
+
_lifecycle_step(
|
|
1171
|
+
"report_replay",
|
|
1172
|
+
"Report Replay",
|
|
1173
|
+
[
|
|
1174
|
+
"agent-learn",
|
|
1175
|
+
"report",
|
|
1176
|
+
paths["replay"],
|
|
1177
|
+
"--output",
|
|
1178
|
+
paths["replay_report"],
|
|
1179
|
+
"--markdown",
|
|
1180
|
+
paths["replay_report_markdown"],
|
|
1181
|
+
],
|
|
1182
|
+
outputs={
|
|
1183
|
+
"json": paths["replay_report"],
|
|
1184
|
+
"markdown": paths["replay_report_markdown"],
|
|
1185
|
+
},
|
|
1186
|
+
),
|
|
1187
|
+
]
|
|
1188
|
+
return {
|
|
1189
|
+
"kind": AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND,
|
|
1190
|
+
"name": str(name),
|
|
1191
|
+
"required_env": _unique_strings(required_env),
|
|
1192
|
+
"artifacts": {key: str(value) for key, value in paths.items()},
|
|
1193
|
+
"steps": steps,
|
|
1194
|
+
"metadata": {
|
|
1195
|
+
"source": "fi.alk.suite.build_optimization_lifecycle_plan",
|
|
1196
|
+
"research_synthesis": (
|
|
1197
|
+
"Deterministic optimization transactions: diagnose/search, "
|
|
1198
|
+
"export, promote, replay, and expose action cards over one "
|
|
1199
|
+
"shared evidence trail."
|
|
1200
|
+
),
|
|
1201
|
+
},
|
|
1202
|
+
}
|
|
1203
|
+
|
|
1204
|
+
|
|
1205
|
+
def run_optimization_lifecycle_file(
|
|
1206
|
+
optimize_manifest_path: str | Path,
|
|
1207
|
+
*,
|
|
1208
|
+
workspace_dir: str | Path | None = None,
|
|
1209
|
+
name: str = "optimization-lifecycle",
|
|
1210
|
+
required_env: Sequence[str] = (),
|
|
1211
|
+
) -> dict[str, Any]:
|
|
1212
|
+
"""Run optimize, report, promote, replay, and report replay via SDK."""
|
|
1213
|
+
|
|
1214
|
+
from fi.alk import optimize, simulate
|
|
1215
|
+
|
|
1216
|
+
plan = build_optimization_lifecycle_plan(
|
|
1217
|
+
optimize_manifest_path=optimize_manifest_path,
|
|
1218
|
+
workspace_dir=workspace_dir,
|
|
1219
|
+
name=name,
|
|
1220
|
+
required_env=required_env,
|
|
1221
|
+
)
|
|
1222
|
+
paths = {key: Path(value) for key, value in plan["artifacts"].items()}
|
|
1223
|
+
outputs_written: list[str] = []
|
|
1224
|
+
|
|
1225
|
+
optimization = optimize.optimize_manifest_file(paths["optimize_manifest"])
|
|
1226
|
+
outputs_written.extend(
|
|
1227
|
+
_write_lifecycle_result_bundle(
|
|
1228
|
+
optimization,
|
|
1229
|
+
json_path=paths["optimization"],
|
|
1230
|
+
junit_path=paths["optimization_junit"],
|
|
1231
|
+
sarif_path=paths["optimization_sarif"],
|
|
1232
|
+
markdown_path=paths["optimization_markdown"],
|
|
1233
|
+
source_path=paths["optimize_manifest"],
|
|
1234
|
+
)
|
|
1235
|
+
)
|
|
1236
|
+
|
|
1237
|
+
optimization_report = simulate.render_report(
|
|
1238
|
+
optimization,
|
|
1239
|
+
source_path=paths["optimization"],
|
|
1240
|
+
)
|
|
1241
|
+
outputs_written.extend(
|
|
1242
|
+
_write_lifecycle_report_bundle(
|
|
1243
|
+
optimization_report,
|
|
1244
|
+
json_path=paths["optimization_report"],
|
|
1245
|
+
markdown_path=paths["optimization_report_markdown"],
|
|
1246
|
+
source_path=paths["optimization"],
|
|
1247
|
+
)
|
|
1248
|
+
)
|
|
1249
|
+
|
|
1250
|
+
promotion = simulate.promote_to_regression(
|
|
1251
|
+
optimization,
|
|
1252
|
+
source_path=paths["optimization"],
|
|
1253
|
+
min_level="note",
|
|
1254
|
+
max_findings=1,
|
|
1255
|
+
required_env=required_env,
|
|
1256
|
+
)
|
|
1257
|
+
outputs_written.append(_write_json(paths["promotion"], promotion))
|
|
1258
|
+
manifest = promotion.get("manifest")
|
|
1259
|
+
if isinstance(manifest, Mapping):
|
|
1260
|
+
outputs_written.append(_write_json(paths["regression_manifest"], manifest))
|
|
1261
|
+
|
|
1262
|
+
promotion_report = simulate.render_report(
|
|
1263
|
+
promotion,
|
|
1264
|
+
source_path=paths["promotion"],
|
|
1265
|
+
)
|
|
1266
|
+
outputs_written.extend(
|
|
1267
|
+
_write_lifecycle_report_bundle(
|
|
1268
|
+
promotion_report,
|
|
1269
|
+
json_path=paths["promotion_report"],
|
|
1270
|
+
markdown_path=paths["promotion_report_markdown"],
|
|
1271
|
+
source_path=paths["promotion"],
|
|
1272
|
+
)
|
|
1273
|
+
)
|
|
1274
|
+
|
|
1275
|
+
replay = simulate.replay_manifests([paths["regression_manifest"]])
|
|
1276
|
+
outputs_written.extend(
|
|
1277
|
+
_write_lifecycle_result_bundle(
|
|
1278
|
+
replay,
|
|
1279
|
+
json_path=paths["replay"],
|
|
1280
|
+
junit_path=paths["replay_junit"],
|
|
1281
|
+
sarif_path=paths["replay_sarif"],
|
|
1282
|
+
markdown_path=paths["replay_markdown"],
|
|
1283
|
+
source_path=paths["regression_manifest"],
|
|
1284
|
+
)
|
|
1285
|
+
)
|
|
1286
|
+
|
|
1287
|
+
replay_report = simulate.render_report(replay, source_path=paths["replay"])
|
|
1288
|
+
outputs_written.extend(
|
|
1289
|
+
_write_lifecycle_report_bundle(
|
|
1290
|
+
replay_report,
|
|
1291
|
+
json_path=paths["replay_report"],
|
|
1292
|
+
markdown_path=paths["replay_report_markdown"],
|
|
1293
|
+
source_path=paths["replay"],
|
|
1294
|
+
)
|
|
1295
|
+
)
|
|
1296
|
+
|
|
1297
|
+
passed = all(
|
|
1298
|
+
payload.get("status") == "passed"
|
|
1299
|
+
for payload in (optimization, promotion, replay)
|
|
1300
|
+
)
|
|
1301
|
+
return {
|
|
1302
|
+
"kind": AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND,
|
|
1303
|
+
"name": str(name),
|
|
1304
|
+
"status": "passed" if passed else "failed",
|
|
1305
|
+
"exit_code": 0 if passed else 1,
|
|
1306
|
+
"summary": {
|
|
1307
|
+
"optimization_score": dict(optimization.get("summary") or {}).get(
|
|
1308
|
+
"optimization_score"
|
|
1309
|
+
),
|
|
1310
|
+
"promotion_kind": dict(promotion.get("summary") or {}).get(
|
|
1311
|
+
"promotion_kind"
|
|
1312
|
+
),
|
|
1313
|
+
"promoted_manifest_count": dict(promotion.get("summary") or {}).get(
|
|
1314
|
+
"promoted_manifest_count"
|
|
1315
|
+
),
|
|
1316
|
+
"replay_pass_rate": dict(replay.get("summary") or {}).get(
|
|
1317
|
+
"replay_pass_rate"
|
|
1318
|
+
),
|
|
1319
|
+
"step_count": len(plan["steps"]),
|
|
1320
|
+
"outputs_written_count": len(outputs_written),
|
|
1321
|
+
},
|
|
1322
|
+
"plan": plan,
|
|
1323
|
+
"artifacts": {
|
|
1324
|
+
"optimization": optimization,
|
|
1325
|
+
"optimization_report": optimization_report,
|
|
1326
|
+
"promotion": promotion,
|
|
1327
|
+
"promotion_report": promotion_report,
|
|
1328
|
+
"replay": replay,
|
|
1329
|
+
"replay_report": replay_report,
|
|
1330
|
+
},
|
|
1331
|
+
"outputs_written": outputs_written,
|
|
1332
|
+
}
|
|
1333
|
+
|
|
1334
|
+
|
|
1335
|
+
def write_suite_file(manifest: Mapping[str, Any], path: str | Path) -> Path:
|
|
1336
|
+
"""Write a suite manifest as formatted JSON and return the resolved path."""
|
|
1337
|
+
|
|
1338
|
+
suite_path = Path(path).expanduser().resolve()
|
|
1339
|
+
suite_path.parent.mkdir(parents=True, exist_ok=True)
|
|
1340
|
+
suite_path.write_text(
|
|
1341
|
+
json.dumps(dict(manifest), indent=2, sort_keys=True, default=str) + "\n",
|
|
1342
|
+
encoding="utf-8",
|
|
1343
|
+
)
|
|
1344
|
+
return suite_path
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _write_suite_json(payload: Mapping[str, Any], path: str | Path) -> Path:
|
|
1348
|
+
output_path = Path(path).expanduser().resolve()
|
|
1349
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
1350
|
+
output_path.write_text(
|
|
1351
|
+
json.dumps(dict(payload), indent=2, sort_keys=True, default=str) + "\n",
|
|
1352
|
+
encoding="utf-8",
|
|
1353
|
+
)
|
|
1354
|
+
return output_path
|
|
1355
|
+
|
|
1356
|
+
|
|
1357
|
+
def run_suite_file(
|
|
1358
|
+
path: str | Path,
|
|
1359
|
+
*,
|
|
1360
|
+
options: Optional[SuiteRunOptions] = None,
|
|
1361
|
+
name: Optional[str] = None,
|
|
1362
|
+
threshold: Optional[float] = None,
|
|
1363
|
+
max_candidates: Optional[int] = None,
|
|
1364
|
+
dry_run: Optional[bool] = None,
|
|
1365
|
+
fail_fast: Optional[bool] = None,
|
|
1366
|
+
require_optimizer_governance: Optional[bool] = None,
|
|
1367
|
+
) -> dict[str, Any]:
|
|
1368
|
+
suite_path = Path(path).expanduser().resolve()
|
|
1369
|
+
suite = load_suite_file(suite_path)
|
|
1370
|
+
return run_suite(
|
|
1371
|
+
suite,
|
|
1372
|
+
suite_path=suite_path,
|
|
1373
|
+
options=_merge_options(
|
|
1374
|
+
options,
|
|
1375
|
+
name=name,
|
|
1376
|
+
threshold=threshold,
|
|
1377
|
+
max_candidates=max_candidates,
|
|
1378
|
+
dry_run=dry_run,
|
|
1379
|
+
fail_fast=fail_fast,
|
|
1380
|
+
require_optimizer_governance=require_optimizer_governance,
|
|
1381
|
+
),
|
|
1382
|
+
)
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
def run_suite(
|
|
1386
|
+
suite: Mapping[str, Any],
|
|
1387
|
+
*,
|
|
1388
|
+
suite_path: str | Path = ".",
|
|
1389
|
+
options: Optional[SuiteRunOptions] = None,
|
|
1390
|
+
name: Optional[str] = None,
|
|
1391
|
+
threshold: Optional[float] = None,
|
|
1392
|
+
max_candidates: Optional[int] = None,
|
|
1393
|
+
dry_run: Optional[bool] = None,
|
|
1394
|
+
fail_fast: Optional[bool] = None,
|
|
1395
|
+
require_optimizer_governance: Optional[bool] = None,
|
|
1396
|
+
) -> dict[str, Any]:
|
|
1397
|
+
started = time.time()
|
|
1398
|
+
opts = _merge_options(
|
|
1399
|
+
options,
|
|
1400
|
+
name=name,
|
|
1401
|
+
threshold=threshold,
|
|
1402
|
+
max_candidates=max_candidates,
|
|
1403
|
+
dry_run=dry_run,
|
|
1404
|
+
fail_fast=fail_fast,
|
|
1405
|
+
require_optimizer_governance=require_optimizer_governance,
|
|
1406
|
+
)
|
|
1407
|
+
suite_path = Path(suite_path).expanduser().resolve()
|
|
1408
|
+
base_dir = _suite_base_dir(suite_path)
|
|
1409
|
+
runtime_suite = _prepare_suite(copy.deepcopy(dict(suite)), base_dir=base_dir)
|
|
1410
|
+
if opts.require_optimizer_governance:
|
|
1411
|
+
optimizer_policy = _suite_optimizer_governance_policy(runtime_suite)
|
|
1412
|
+
optimizer_policy["require_optimizer_governance"] = True
|
|
1413
|
+
optimizer_policy["require_passed"] = True
|
|
1414
|
+
optimizer_policy["min_governed"] = max(
|
|
1415
|
+
int(optimizer_policy.get("min_governed") or 0),
|
|
1416
|
+
1,
|
|
1417
|
+
)
|
|
1418
|
+
runtime_suite["optimizer_governance_policy"] = {
|
|
1419
|
+
**optimizer_policy,
|
|
1420
|
+
}
|
|
1421
|
+
validate_suite_env(runtime_suite, suite_path=suite_path)
|
|
1422
|
+
|
|
1423
|
+
children: list[dict[str, Any]] = []
|
|
1424
|
+
for index, job in enumerate(_suite_jobs(runtime_suite), start=1):
|
|
1425
|
+
child = _execute_job(
|
|
1426
|
+
job,
|
|
1427
|
+
index=index,
|
|
1428
|
+
base_dir=base_dir,
|
|
1429
|
+
suite_options=opts,
|
|
1430
|
+
)
|
|
1431
|
+
children.append(child)
|
|
1432
|
+
if int(child.get("exit_code", 1)) != 0 and opts.fail_fast:
|
|
1433
|
+
break
|
|
1434
|
+
|
|
1435
|
+
payload = _suite_result(
|
|
1436
|
+
suite=runtime_suite,
|
|
1437
|
+
suite_path=suite_path,
|
|
1438
|
+
children=children,
|
|
1439
|
+
name=opts.name,
|
|
1440
|
+
dry_run=opts.dry_run,
|
|
1441
|
+
fail_fast=opts.fail_fast,
|
|
1442
|
+
duration_seconds=round(time.time() - started, 4),
|
|
1443
|
+
)
|
|
1444
|
+
return public_payload(payload, kind=AGENT_LEARNING_SUITE_KIND)
|
|
1445
|
+
|
|
1446
|
+
|
|
1447
|
+
def optimize_suite_file(
|
|
1448
|
+
path: str | Path,
|
|
1449
|
+
*,
|
|
1450
|
+
options: Optional[SuiteOptimizationOptions] = None,
|
|
1451
|
+
name: Optional[str] = None,
|
|
1452
|
+
threshold: Optional[float] = None,
|
|
1453
|
+
max_candidates: Optional[int] = None,
|
|
1454
|
+
dry_run: Optional[bool] = None,
|
|
1455
|
+
) -> dict[str, Any]:
|
|
1456
|
+
"""Load and optimize a full Agent Learning suite."""
|
|
1457
|
+
|
|
1458
|
+
suite_path = Path(path).expanduser().resolve()
|
|
1459
|
+
suite = load_suite_file(suite_path)
|
|
1460
|
+
return optimize_suite(
|
|
1461
|
+
suite,
|
|
1462
|
+
suite_path=suite_path,
|
|
1463
|
+
options=_merge_optimization_options(
|
|
1464
|
+
options,
|
|
1465
|
+
name=name,
|
|
1466
|
+
threshold=threshold,
|
|
1467
|
+
max_candidates=max_candidates,
|
|
1468
|
+
dry_run=dry_run,
|
|
1469
|
+
),
|
|
1470
|
+
)
|
|
1471
|
+
|
|
1472
|
+
|
|
1473
|
+
def optimize_suite(
|
|
1474
|
+
suite: Mapping[str, Any],
|
|
1475
|
+
*,
|
|
1476
|
+
suite_path: str | Path = ".",
|
|
1477
|
+
options: Optional[SuiteOptimizationOptions] = None,
|
|
1478
|
+
name: Optional[str] = None,
|
|
1479
|
+
threshold: Optional[float] = None,
|
|
1480
|
+
max_candidates: Optional[int] = None,
|
|
1481
|
+
dry_run: Optional[bool] = None,
|
|
1482
|
+
) -> dict[str, Any]:
|
|
1483
|
+
"""Optimize a mixed Agent Learning suite and return a unified artifact."""
|
|
1484
|
+
|
|
1485
|
+
started = time.time()
|
|
1486
|
+
opts = _merge_optimization_options(
|
|
1487
|
+
options,
|
|
1488
|
+
name=name,
|
|
1489
|
+
threshold=threshold,
|
|
1490
|
+
max_candidates=max_candidates,
|
|
1491
|
+
dry_run=dry_run,
|
|
1492
|
+
)
|
|
1493
|
+
suite_path = Path(suite_path).expanduser().resolve()
|
|
1494
|
+
base_dir = _suite_base_dir(suite_path)
|
|
1495
|
+
runtime_suite = copy.deepcopy(dict(suite))
|
|
1496
|
+
if opts.name:
|
|
1497
|
+
runtime_suite["name"] = opts.name
|
|
1498
|
+
if opts.threshold is not None:
|
|
1499
|
+
runtime_suite.setdefault("optimization", {})["threshold"] = opts.threshold
|
|
1500
|
+
if opts.max_candidates is not None:
|
|
1501
|
+
runtime_suite.setdefault("optimization", {}).setdefault(
|
|
1502
|
+
"optimizer", {}
|
|
1503
|
+
)["max_candidates"] = opts.max_candidates
|
|
1504
|
+
|
|
1505
|
+
prepared = _prepare_suite(runtime_suite, base_dir=base_dir)
|
|
1506
|
+
validate_suite_env(prepared, suite_path=suite_path)
|
|
1507
|
+
cli = _optimization_cli()
|
|
1508
|
+
optimization = cli._optimization_config(prepared)
|
|
1509
|
+
target_config = cli._target_config(optimization)
|
|
1510
|
+
optimizer_config = cli._optimizer_config(optimization)
|
|
1511
|
+
if opts.dry_run:
|
|
1512
|
+
return public_payload({
|
|
1513
|
+
"schema_version": AGENT_LEARNING_CLI_SCHEMA_VERSION,
|
|
1514
|
+
"kind": AGENT_LEARNING_SUITE_OPTIMIZATION_KIND,
|
|
1515
|
+
"name": str(prepared.get("name") or suite_path.stem),
|
|
1516
|
+
"status": "passed",
|
|
1517
|
+
"exit_code": 0,
|
|
1518
|
+
"dry_run": True,
|
|
1519
|
+
"summary": {
|
|
1520
|
+
"job_count": len(_suite_jobs(prepared)),
|
|
1521
|
+
"required_env": required_suite_env(prepared, suite_path=suite_path),
|
|
1522
|
+
"search_path_count": len(target_config.get("search_space", {})),
|
|
1523
|
+
"max_candidates": optimizer_config.get("max_candidates"),
|
|
1524
|
+
},
|
|
1525
|
+
"duration_seconds": round(time.time() - started, 4),
|
|
1526
|
+
}, kind=AGENT_LEARNING_SUITE_OPTIMIZATION_KIND)
|
|
1527
|
+
|
|
1528
|
+
try:
|
|
1529
|
+
from fi.alk import optimize as agent_optimize
|
|
1530
|
+
except Exception as exc: # pragma: no cover - optional dependency clarity
|
|
1531
|
+
raise SuiteError(
|
|
1532
|
+
"Agent Learning Kit optimizer engine is required for suite optimization."
|
|
1533
|
+
) from exc
|
|
1534
|
+
|
|
1535
|
+
problem = agent_optimize.problem_from_agent_learning_suite(
|
|
1536
|
+
prepared,
|
|
1537
|
+
suite_path=suite_path,
|
|
1538
|
+
name=str(prepared.get("name") or suite_path.stem),
|
|
1539
|
+
)
|
|
1540
|
+
optimization_result = problem.optimize()
|
|
1541
|
+
payload = cli._optimization_result(
|
|
1542
|
+
manifest=prepared,
|
|
1543
|
+
manifest_path=suite_path,
|
|
1544
|
+
optimization_result=optimization_result,
|
|
1545
|
+
threshold=float(optimization.get("threshold", 1.0)),
|
|
1546
|
+
duration_seconds=round(time.time() - started, 4),
|
|
1547
|
+
)
|
|
1548
|
+
payload["kind"] = AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
|
|
1549
|
+
payload["suite"] = _suite_descriptor(prepared)
|
|
1550
|
+
payload["optimization"]["source"] = "agent_learning_suite"
|
|
1551
|
+
if "manifest_optimization" in payload["optimization"]:
|
|
1552
|
+
artifact = copy.deepcopy(payload["optimization"]["manifest_optimization"])
|
|
1553
|
+
artifact["kind"] = "agent_learning_suite_optimization"
|
|
1554
|
+
artifact["source"] = "agent_learning_suite"
|
|
1555
|
+
payload["optimization"]["suite_optimization"] = artifact
|
|
1556
|
+
payload["summary"]["job_count"] = len(_suite_jobs(prepared))
|
|
1557
|
+
payload["summary"]["child_command_count"] = _suite_job_command_counts(prepared)
|
|
1558
|
+
action_plan = _artifact_action_plan_card(payload)
|
|
1559
|
+
if action_plan is not None:
|
|
1560
|
+
payload["artifact_action_plan"] = action_plan
|
|
1561
|
+
payload["optimization"]["artifact_action_plan"] = copy.deepcopy(action_plan)
|
|
1562
|
+
payload["summary"]["artifact_action_best_action_id"] = action_plan.get(
|
|
1563
|
+
"selected_action_id"
|
|
1564
|
+
)
|
|
1565
|
+
return public_payload(payload, kind=AGENT_LEARNING_SUITE_OPTIMIZATION_KIND)
|
|
1566
|
+
|
|
1567
|
+
|
|
1568
|
+
def render_junit(result: Mapping[str, Any]) -> str:
|
|
1569
|
+
name = escape(str(result.get("name") or "agent-learning-suite"))
|
|
1570
|
+
children = list(result.get("children") or result.get("jobs") or [])
|
|
1571
|
+
finding_failures = [
|
|
1572
|
+
finding
|
|
1573
|
+
for finding in list(result.get("findings") or [])
|
|
1574
|
+
if str(_as_mapping(finding).get("type"))
|
|
1575
|
+
in {
|
|
1576
|
+
"suite_required_capability_missing",
|
|
1577
|
+
"suite_evidence_admission_missing",
|
|
1578
|
+
"suite_evidence_freeze_missing",
|
|
1579
|
+
"suite_framework_adapter_conformance_failed",
|
|
1580
|
+
"suite_framework_coverage_missing",
|
|
1581
|
+
"suite_optimizer_governance_failed",
|
|
1582
|
+
"suite_optimizer_governance_missing",
|
|
1583
|
+
"suite_optimizer_governance_warning",
|
|
1584
|
+
}
|
|
1585
|
+
]
|
|
1586
|
+
failures = (
|
|
1587
|
+
sum(1 for child in children if int(child.get("exit_code", 1)) != 0)
|
|
1588
|
+
+ len(finding_failures)
|
|
1589
|
+
)
|
|
1590
|
+
lines = [
|
|
1591
|
+
(
|
|
1592
|
+
f'<testsuite name="{name}" tests="{len(children) + len(finding_failures)}" '
|
|
1593
|
+
f'failures="{failures}" errors="0">'
|
|
1594
|
+
)
|
|
1595
|
+
]
|
|
1596
|
+
for child in children:
|
|
1597
|
+
child_name = escape(str(child.get("id") or child.get("name") or "job"))
|
|
1598
|
+
class_name = escape(str(child.get("command") or "suite"))
|
|
1599
|
+
duration = float(child.get("duration_seconds") or 0.0)
|
|
1600
|
+
lines.append(
|
|
1601
|
+
f' <testcase classname="{class_name}" name="{child_name}" '
|
|
1602
|
+
f'time="{duration:.4f}">'
|
|
1603
|
+
)
|
|
1604
|
+
if int(child.get("exit_code", 1)) != 0:
|
|
1605
|
+
message = escape(str(child.get("error") or child.get("status") or "failed"))
|
|
1606
|
+
lines.append(f' <failure message="{message}">{message}</failure>')
|
|
1607
|
+
lines.append(" </testcase>")
|
|
1608
|
+
for index, finding in enumerate(finding_failures, start=1):
|
|
1609
|
+
item = _as_mapping(finding)
|
|
1610
|
+
finding_name = escape(str(item.get("type") or f"suite_finding_{index}"))
|
|
1611
|
+
message = escape(str(item.get("reason") or finding_name))
|
|
1612
|
+
lines.append(f' <testcase classname="suite" name="{finding_name}" time="0.0000">')
|
|
1613
|
+
lines.append(f' <failure message="{message}">{message}</failure>')
|
|
1614
|
+
lines.append(" </testcase>")
|
|
1615
|
+
lines.append("</testsuite>")
|
|
1616
|
+
return "\n".join(lines)
|
|
1617
|
+
|
|
1618
|
+
|
|
1619
|
+
def render_sarif(
|
|
1620
|
+
result: Mapping[str, Any],
|
|
1621
|
+
*,
|
|
1622
|
+
manifest_path: str | Path = ".",
|
|
1623
|
+
) -> str:
|
|
1624
|
+
suite_path = Path(manifest_path).expanduser().resolve()
|
|
1625
|
+
findings = _suite_sarif_findings(result)
|
|
1626
|
+
sarif_results = []
|
|
1627
|
+
for finding in findings:
|
|
1628
|
+
rule_id = str(finding.get("type") or finding.get("rule_id") or "suite_finding")
|
|
1629
|
+
level = str(finding.get("level") or finding.get("severity") or "error").lower()
|
|
1630
|
+
if level not in {"none", "note", "warning", "error"}:
|
|
1631
|
+
level = "warning"
|
|
1632
|
+
location_path = str(finding.get("path") or suite_path)
|
|
1633
|
+
sarif_results.append(
|
|
1634
|
+
{
|
|
1635
|
+
"ruleId": rule_id,
|
|
1636
|
+
"level": level,
|
|
1637
|
+
"message": {"text": str(finding.get("reason") or rule_id)},
|
|
1638
|
+
"locations": [
|
|
1639
|
+
{
|
|
1640
|
+
"physicalLocation": {
|
|
1641
|
+
"artifactLocation": {"uri": location_path},
|
|
1642
|
+
}
|
|
1643
|
+
}
|
|
1644
|
+
],
|
|
1645
|
+
}
|
|
1646
|
+
)
|
|
1647
|
+
payload = {
|
|
1648
|
+
"$schema": "https://json.schemastore.org/sarif-2.1.0.json",
|
|
1649
|
+
"version": "2.1.0",
|
|
1650
|
+
"runs": [
|
|
1651
|
+
{
|
|
1652
|
+
"tool": {
|
|
1653
|
+
"driver": {
|
|
1654
|
+
"name": "agent-learning-suite",
|
|
1655
|
+
"informationUri": "https://futureagi.com",
|
|
1656
|
+
"rules": [],
|
|
1657
|
+
}
|
|
1658
|
+
},
|
|
1659
|
+
"results": sarif_results,
|
|
1660
|
+
}
|
|
1661
|
+
],
|
|
1662
|
+
}
|
|
1663
|
+
return json.dumps(payload, indent=2, sort_keys=True)
|
|
1664
|
+
|
|
1665
|
+
|
|
1666
|
+
def render_markdown(
|
|
1667
|
+
result: Mapping[str, Any],
|
|
1668
|
+
*,
|
|
1669
|
+
source_path: str | Path = ".",
|
|
1670
|
+
) -> str:
|
|
1671
|
+
summary = dict(result.get("summary") or {})
|
|
1672
|
+
certificate = _as_mapping(result.get("trust_certificate"))
|
|
1673
|
+
lines = [
|
|
1674
|
+
f"# {result.get('name') or 'agent-learning-suite'}",
|
|
1675
|
+
"",
|
|
1676
|
+
f"- Source: `{Path(source_path)}`",
|
|
1677
|
+
f"- Status: `{result.get('status')}`",
|
|
1678
|
+
f"- Jobs: {summary.get('passed_count', 0)}/{summary.get('job_count', 0)} passed",
|
|
1679
|
+
f"- Score: {summary.get('score', 0.0)}",
|
|
1680
|
+
(
|
|
1681
|
+
"- Trust Certificate: "
|
|
1682
|
+
f"{certificate.get('verdict') or summary.get('trust_certificate_verdict')}"
|
|
1683
|
+
f" ({certificate.get('assurance_level') or summary.get('trust_certificate_assurance_level')})"
|
|
1684
|
+
),
|
|
1685
|
+
(
|
|
1686
|
+
"- Evidence: "
|
|
1687
|
+
f"{summary.get('admitted_evidence_count', 0)} admitted, "
|
|
1688
|
+
f"{summary.get('non_admitted_evidence_count', 0)} non-admitted, "
|
|
1689
|
+
f"{summary.get('rejected_evidence_count', 0)} rejected, "
|
|
1690
|
+
f"{summary.get('frozen_evidence_count', 0)} frozen"
|
|
1691
|
+
),
|
|
1692
|
+
(
|
|
1693
|
+
"- Frameworks: "
|
|
1694
|
+
f"{summary.get('observed_framework_count', 0)} observed, "
|
|
1695
|
+
f"{summary.get('missing_framework_count', 0)} missing, "
|
|
1696
|
+
f"{summary.get('adapter_conformance_failed_count', 0)} adapter-failed"
|
|
1697
|
+
),
|
|
1698
|
+
"",
|
|
1699
|
+
"## Trust Certificate",
|
|
1700
|
+
"",
|
|
1701
|
+
f"- Verdict: `{certificate.get('verdict')}`",
|
|
1702
|
+
f"- Assurance Level: `{certificate.get('assurance_level')}`",
|
|
1703
|
+
f"- Promotion Ready: `{certificate.get('promotion_ready')}`",
|
|
1704
|
+
f"- Reason: {certificate.get('reason') or ''}",
|
|
1705
|
+
"",
|
|
1706
|
+
"| Gate | Status | Required |",
|
|
1707
|
+
"| --- | --- | --- |",
|
|
1708
|
+
]
|
|
1709
|
+
for gate in _as_list(certificate.get("gates")):
|
|
1710
|
+
gate_item = _as_mapping(gate)
|
|
1711
|
+
if not gate_item:
|
|
1712
|
+
continue
|
|
1713
|
+
lines.append(
|
|
1714
|
+
"| "
|
|
1715
|
+
f"{_md_cell(gate_item.get('id') or '')} | "
|
|
1716
|
+
f"{_md_cell(gate_item.get('status') or '')} | "
|
|
1717
|
+
f"{_md_cell(str(bool(gate_item.get('required'))))} |"
|
|
1718
|
+
)
|
|
1719
|
+
lines.extend([
|
|
1720
|
+
"",
|
|
1721
|
+
"| Job | Command | Status | Evidence | Exit |",
|
|
1722
|
+
"| --- | --- | --- | --- | --- |",
|
|
1723
|
+
])
|
|
1724
|
+
for child in list(result.get("children") or result.get("jobs") or []):
|
|
1725
|
+
evidence = _as_mapping(child.get("evidence"))
|
|
1726
|
+
evidence_cell = evidence.get("status") or ""
|
|
1727
|
+
if evidence.get("role") and evidence.get("role") != evidence_cell:
|
|
1728
|
+
evidence_cell = f"{evidence_cell} ({evidence.get('role')})"
|
|
1729
|
+
lines.append(
|
|
1730
|
+
"| "
|
|
1731
|
+
f"{_md_cell(child.get('id') or child.get('name') or '')} | "
|
|
1732
|
+
f"{_md_cell(child.get('command') or '')} | "
|
|
1733
|
+
f"{_md_cell(child.get('status') or '')} | "
|
|
1734
|
+
f"{_md_cell(evidence_cell)} | "
|
|
1735
|
+
f"{int(child.get('exit_code', 1))} |"
|
|
1736
|
+
)
|
|
1737
|
+
return "\n".join(lines) + "\n"
|
|
1738
|
+
|
|
1739
|
+
|
|
1740
|
+
def _prepare_suite(suite: dict[str, Any], *, base_dir: Path) -> dict[str, Any]:
|
|
1741
|
+
jobs = _as_list(suite.get("jobs") or suite.get("runs") or suite.get("steps"))
|
|
1742
|
+
if not jobs:
|
|
1743
|
+
raise SuiteError("suite manifest requires at least one job")
|
|
1744
|
+
prepared_jobs = []
|
|
1745
|
+
for index, job in enumerate(jobs, start=1):
|
|
1746
|
+
if not isinstance(job, Mapping):
|
|
1747
|
+
raise SuiteError(f"suite job[{index}] must be an object")
|
|
1748
|
+
prepared = dict(job)
|
|
1749
|
+
prepared["command"] = _normalize_command(
|
|
1750
|
+
prepared.get("command") or prepared.get("type") or prepared.get("kind")
|
|
1751
|
+
)
|
|
1752
|
+
prepared.setdefault("id", f"{prepared['command']}-{index}")
|
|
1753
|
+
_job_path(prepared, base_dir=base_dir)
|
|
1754
|
+
prepared_jobs.append(prepared)
|
|
1755
|
+
suite["jobs"] = prepared_jobs
|
|
1756
|
+
suite.setdefault("version", AGENT_LEARNING_SUITE_KIND)
|
|
1757
|
+
suite.setdefault("name", "agent-learning-suite")
|
|
1758
|
+
return suite
|
|
1759
|
+
|
|
1760
|
+
|
|
1761
|
+
def _execute_job(
|
|
1762
|
+
job: Mapping[str, Any],
|
|
1763
|
+
*,
|
|
1764
|
+
index: int,
|
|
1765
|
+
base_dir: Path,
|
|
1766
|
+
suite_options: SuiteRunOptions,
|
|
1767
|
+
) -> dict[str, Any]:
|
|
1768
|
+
started = time.time()
|
|
1769
|
+
command = _normalize_command(job.get("command") or job.get("type"))
|
|
1770
|
+
path = _job_path(job, base_dir=base_dir)
|
|
1771
|
+
job_id = str(job.get("id") or f"{command}-{index}")
|
|
1772
|
+
try:
|
|
1773
|
+
payload = _execute_child_payload(
|
|
1774
|
+
command,
|
|
1775
|
+
path=path,
|
|
1776
|
+
base_dir=base_dir,
|
|
1777
|
+
job=job,
|
|
1778
|
+
suite_options=suite_options,
|
|
1779
|
+
)
|
|
1780
|
+
payload = copy.deepcopy(dict(payload))
|
|
1781
|
+
outputs_written = _write_child_outputs(
|
|
1782
|
+
payload,
|
|
1783
|
+
command=command,
|
|
1784
|
+
job=job,
|
|
1785
|
+
path=path,
|
|
1786
|
+
)
|
|
1787
|
+
payload["outputs_written"] = outputs_written
|
|
1788
|
+
result = {
|
|
1789
|
+
"id": job_id,
|
|
1790
|
+
"command": command,
|
|
1791
|
+
"path": str(path),
|
|
1792
|
+
"kind": payload.get("kind"),
|
|
1793
|
+
"name": payload.get("name"),
|
|
1794
|
+
"status": str(payload.get("status") or "unknown"),
|
|
1795
|
+
"exit_code": int(payload.get("exit_code", 1)),
|
|
1796
|
+
"summary": copy.deepcopy(dict(payload.get("summary") or {})),
|
|
1797
|
+
"findings": copy.deepcopy(list(payload.get("findings") or [])),
|
|
1798
|
+
"outputs_written": outputs_written,
|
|
1799
|
+
"duration_seconds": round(time.time() - started, 4),
|
|
1800
|
+
"result": payload,
|
|
1801
|
+
}
|
|
1802
|
+
result["evidence"] = _suite_child_evidence(
|
|
1803
|
+
job,
|
|
1804
|
+
result,
|
|
1805
|
+
base_dir=base_dir,
|
|
1806
|
+
)
|
|
1807
|
+
return result
|
|
1808
|
+
except Exception as exc:
|
|
1809
|
+
result = {
|
|
1810
|
+
"id": job_id,
|
|
1811
|
+
"command": command,
|
|
1812
|
+
"path": str(path),
|
|
1813
|
+
"kind": None,
|
|
1814
|
+
"name": job.get("name"),
|
|
1815
|
+
"status": "failed",
|
|
1816
|
+
"exit_code": 1,
|
|
1817
|
+
"summary": {},
|
|
1818
|
+
"findings": [
|
|
1819
|
+
{
|
|
1820
|
+
"type": "suite_child_failed",
|
|
1821
|
+
"level": "error",
|
|
1822
|
+
"reason": str(exc),
|
|
1823
|
+
"job": job_id,
|
|
1824
|
+
"command": command,
|
|
1825
|
+
"path": str(path),
|
|
1826
|
+
}
|
|
1827
|
+
],
|
|
1828
|
+
"outputs_written": [],
|
|
1829
|
+
"duration_seconds": round(time.time() - started, 4),
|
|
1830
|
+
"error": str(exc),
|
|
1831
|
+
}
|
|
1832
|
+
result["evidence"] = _suite_child_evidence(
|
|
1833
|
+
job,
|
|
1834
|
+
result,
|
|
1835
|
+
base_dir=base_dir,
|
|
1836
|
+
)
|
|
1837
|
+
return result
|
|
1838
|
+
|
|
1839
|
+
|
|
1840
|
+
def _execute_child_payload(
|
|
1841
|
+
command: str,
|
|
1842
|
+
*,
|
|
1843
|
+
path: Path,
|
|
1844
|
+
base_dir: Path,
|
|
1845
|
+
job: Mapping[str, Any],
|
|
1846
|
+
suite_options: SuiteRunOptions,
|
|
1847
|
+
) -> dict[str, Any]:
|
|
1848
|
+
if command == "run":
|
|
1849
|
+
from fi.alk import simulate
|
|
1850
|
+
from fi.alk.cli import AGENT_LEARNING_RUN_KIND
|
|
1851
|
+
|
|
1852
|
+
payload = _run_async(
|
|
1853
|
+
simulate.run_manifest_file(
|
|
1854
|
+
path,
|
|
1855
|
+
name=_job_name(job),
|
|
1856
|
+
threshold=_job_threshold(job, suite_options),
|
|
1857
|
+
no_eval=bool(job.get("no_eval", job.get("no-eval", False))),
|
|
1858
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1859
|
+
)
|
|
1860
|
+
)
|
|
1861
|
+
payload["kind"] = AGENT_LEARNING_RUN_KIND
|
|
1862
|
+
return payload
|
|
1863
|
+
if command == "suite":
|
|
1864
|
+
payload = run_suite_file(
|
|
1865
|
+
path,
|
|
1866
|
+
options=SuiteRunOptions(
|
|
1867
|
+
name=_job_name(job),
|
|
1868
|
+
threshold=_job_threshold(job, suite_options),
|
|
1869
|
+
max_candidates=_job_max_candidates(job, suite_options),
|
|
1870
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1871
|
+
fail_fast=bool(
|
|
1872
|
+
suite_options.fail_fast
|
|
1873
|
+
or job.get("fail_fast")
|
|
1874
|
+
or job.get("fail-fast")
|
|
1875
|
+
),
|
|
1876
|
+
require_optimizer_governance=suite_options.require_optimizer_governance,
|
|
1877
|
+
),
|
|
1878
|
+
)
|
|
1879
|
+
payload["kind"] = AGENT_LEARNING_SUITE_KIND
|
|
1880
|
+
return payload
|
|
1881
|
+
if command == "action_run":
|
|
1882
|
+
from fi.alk import actions
|
|
1883
|
+
|
|
1884
|
+
artifact = actions.load_artifact_file(path)
|
|
1885
|
+
return actions.run_action(
|
|
1886
|
+
artifact,
|
|
1887
|
+
_job_action_id(job),
|
|
1888
|
+
source_path=path,
|
|
1889
|
+
inputs=_job_action_inputs(job),
|
|
1890
|
+
cwd=_job_action_cwd(job, base_dir=base_dir),
|
|
1891
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1892
|
+
name=_job_name(job),
|
|
1893
|
+
artifact_output_path=_job_action_artifact_output(job),
|
|
1894
|
+
)
|
|
1895
|
+
if command == "eval":
|
|
1896
|
+
from fi.alk import evals
|
|
1897
|
+
from fi.alk.cli import AGENT_LEARNING_EVAL_KIND
|
|
1898
|
+
|
|
1899
|
+
payload = evals.run_eval_suite_file(
|
|
1900
|
+
path,
|
|
1901
|
+
name=_job_name(job),
|
|
1902
|
+
threshold=_job_threshold(job, suite_options),
|
|
1903
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1904
|
+
)
|
|
1905
|
+
payload["kind"] = AGENT_LEARNING_EVAL_KIND
|
|
1906
|
+
return payload
|
|
1907
|
+
if command == "eval_artifact":
|
|
1908
|
+
from fi.alk import evals
|
|
1909
|
+
from fi.alk.cli import AGENT_LEARNING_ARTIFACT_EVAL_KIND
|
|
1910
|
+
|
|
1911
|
+
config_path = _job_optional_path(
|
|
1912
|
+
job,
|
|
1913
|
+
base_dir=base_dir,
|
|
1914
|
+
keys=("config", "eval_config", "agent_report_config"),
|
|
1915
|
+
)
|
|
1916
|
+
config = evals.load_artifact_file(config_path) if config_path else None
|
|
1917
|
+
payload = evals.evaluate_artifact_file(
|
|
1918
|
+
path,
|
|
1919
|
+
config=config,
|
|
1920
|
+
name=_job_name(job),
|
|
1921
|
+
threshold=float(_job_threshold(job, suite_options) or 0.7),
|
|
1922
|
+
)
|
|
1923
|
+
payload["kind"] = AGENT_LEARNING_ARTIFACT_EVAL_KIND
|
|
1924
|
+
return payload
|
|
1925
|
+
if command == "eval_task":
|
|
1926
|
+
from fi.alk import evals
|
|
1927
|
+
from fi.alk.cli import AGENT_LEARNING_ARTIFACT_EVAL_KIND
|
|
1928
|
+
|
|
1929
|
+
config_path = _job_optional_path(
|
|
1930
|
+
job,
|
|
1931
|
+
base_dir=base_dir,
|
|
1932
|
+
keys=("config", "eval_config", "agent_report_config"),
|
|
1933
|
+
)
|
|
1934
|
+
config = evals.load_artifact_file(config_path) if config_path else None
|
|
1935
|
+
payload = evals.evaluate_task_evidence_file(
|
|
1936
|
+
path,
|
|
1937
|
+
config=config,
|
|
1938
|
+
name=_job_name(job),
|
|
1939
|
+
threshold=float(_job_threshold(job, suite_options) or 0.7),
|
|
1940
|
+
)
|
|
1941
|
+
payload["kind"] = AGENT_LEARNING_ARTIFACT_EVAL_KIND
|
|
1942
|
+
return payload
|
|
1943
|
+
if command == "redteam":
|
|
1944
|
+
from fi.alk import redteam
|
|
1945
|
+
|
|
1946
|
+
payload = _run_async(
|
|
1947
|
+
redteam.redteam_manifest_file(
|
|
1948
|
+
path,
|
|
1949
|
+
name=_job_name(job),
|
|
1950
|
+
threshold=_job_threshold(job, suite_options),
|
|
1951
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1952
|
+
)
|
|
1953
|
+
)
|
|
1954
|
+
return payload
|
|
1955
|
+
if command == "optimize":
|
|
1956
|
+
from fi.alk import optimize
|
|
1957
|
+
from fi.alk.cli import AGENT_LEARNING_OPTIMIZATION_KIND
|
|
1958
|
+
|
|
1959
|
+
payload = optimize.optimize_manifest_file(
|
|
1960
|
+
path,
|
|
1961
|
+
name=_job_name(job),
|
|
1962
|
+
threshold=_job_threshold(job, suite_options),
|
|
1963
|
+
max_candidates=_job_max_candidates(job, suite_options),
|
|
1964
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1965
|
+
)
|
|
1966
|
+
payload["kind"] = AGENT_LEARNING_OPTIMIZATION_KIND
|
|
1967
|
+
return payload
|
|
1968
|
+
if command == "optimize_eval":
|
|
1969
|
+
from fi.alk import optimize
|
|
1970
|
+
from fi.alk.cli import AGENT_LEARNING_EVAL_OPTIMIZATION_KIND
|
|
1971
|
+
|
|
1972
|
+
payload = optimize.optimize_eval_suite_file(
|
|
1973
|
+
path,
|
|
1974
|
+
name=_job_name(job),
|
|
1975
|
+
threshold=_job_threshold(job, suite_options),
|
|
1976
|
+
max_candidates=_job_max_candidates(job, suite_options),
|
|
1977
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1978
|
+
)
|
|
1979
|
+
payload["kind"] = AGENT_LEARNING_EVAL_OPTIMIZATION_KIND
|
|
1980
|
+
return payload
|
|
1981
|
+
if command == "optimize_suite":
|
|
1982
|
+
from fi.alk import optimize
|
|
1983
|
+
from fi.alk.cli import AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
|
|
1984
|
+
|
|
1985
|
+
payload = optimize.optimize_suite_file(
|
|
1986
|
+
path,
|
|
1987
|
+
name=_job_name(job),
|
|
1988
|
+
threshold=_job_threshold(job, suite_options),
|
|
1989
|
+
max_candidates=_job_max_candidates(job, suite_options),
|
|
1990
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
1991
|
+
)
|
|
1992
|
+
payload["kind"] = AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
|
|
1993
|
+
return payload
|
|
1994
|
+
if command == "baseline":
|
|
1995
|
+
from fi.alk import simulate
|
|
1996
|
+
|
|
1997
|
+
return simulate.create_baseline_file(
|
|
1998
|
+
path,
|
|
1999
|
+
name=_job_name(job),
|
|
2000
|
+
)
|
|
2001
|
+
if command == "compare":
|
|
2002
|
+
from fi.alk import simulate
|
|
2003
|
+
|
|
2004
|
+
return simulate.compare_result_files(
|
|
2005
|
+
_job_compare_baseline_path(job, base_dir=base_dir),
|
|
2006
|
+
path,
|
|
2007
|
+
min_score_delta=_job_float(job, "min_score_delta", "min-score-delta", default=0.0),
|
|
2008
|
+
max_new_findings=_job_int(job, "max_new_findings", "max-new-findings", default=0),
|
|
2009
|
+
max_new_error_findings=_job_int(
|
|
2010
|
+
job,
|
|
2011
|
+
"max_new_error_findings",
|
|
2012
|
+
"max-new-error-findings",
|
|
2013
|
+
default=0,
|
|
2014
|
+
),
|
|
2015
|
+
min_metric_delta=_job_optional_float(
|
|
2016
|
+
job,
|
|
2017
|
+
"min_metric_delta",
|
|
2018
|
+
"min-metric-delta",
|
|
2019
|
+
),
|
|
2020
|
+
name=_job_name(job),
|
|
2021
|
+
)
|
|
2022
|
+
if command == "report":
|
|
2023
|
+
from fi.alk import simulate
|
|
2024
|
+
|
|
2025
|
+
return simulate.render_report_file(
|
|
2026
|
+
path,
|
|
2027
|
+
name=_job_name(job),
|
|
2028
|
+
)
|
|
2029
|
+
if command == "promote_to_regression":
|
|
2030
|
+
from fi.alk import simulate
|
|
2031
|
+
|
|
2032
|
+
return simulate.promote_to_regression_file(
|
|
2033
|
+
path,
|
|
2034
|
+
name=_job_name(job),
|
|
2035
|
+
min_level=str(job.get("min_level") or job.get("min-level") or "warning"),
|
|
2036
|
+
max_findings=_job_int(job, "max_findings", "max-findings", default=25),
|
|
2037
|
+
required_env=_as_string_list(job.get("required_env")),
|
|
2038
|
+
)
|
|
2039
|
+
if command == "shrink":
|
|
2040
|
+
from fi.alk import simulate
|
|
2041
|
+
|
|
2042
|
+
return simulate.shrink_attack_evolution_file(
|
|
2043
|
+
path,
|
|
2044
|
+
name=_job_name(job),
|
|
2045
|
+
manifest_name=str(
|
|
2046
|
+
job.get("manifest_name")
|
|
2047
|
+
or job.get("manifest-name")
|
|
2048
|
+
or ""
|
|
2049
|
+
)
|
|
2050
|
+
or None,
|
|
2051
|
+
required_env=_as_string_list(job.get("required_env")),
|
|
2052
|
+
)
|
|
2053
|
+
if command == "replay":
|
|
2054
|
+
from fi.alk import simulate
|
|
2055
|
+
|
|
2056
|
+
return simulate.replay_manifests(
|
|
2057
|
+
_job_replay_manifest_paths(job, base_dir=base_dir),
|
|
2058
|
+
name=_job_name(job),
|
|
2059
|
+
dry_run=_job_dry_run(job, suite_options),
|
|
2060
|
+
fail_fast=bool(suite_options.fail_fast or job.get("fail_fast") or job.get("fail-fast")),
|
|
2061
|
+
)
|
|
2062
|
+
raise SuiteError(f"unsupported suite job command: {command}")
|
|
2063
|
+
|
|
2064
|
+
|
|
2065
|
+
def _write_child_outputs(
|
|
2066
|
+
payload: Mapping[str, Any],
|
|
2067
|
+
*,
|
|
2068
|
+
command: str,
|
|
2069
|
+
job: Mapping[str, Any],
|
|
2070
|
+
path: Path,
|
|
2071
|
+
) -> list[str]:
|
|
2072
|
+
output_paths = _job_output_paths(job, path.parent)
|
|
2073
|
+
if not any(output_paths.values()):
|
|
2074
|
+
return []
|
|
2075
|
+
render_junit_fn, render_sarif_fn, render_markdown_fn = _child_renderers(command)
|
|
2076
|
+
written: list[str] = []
|
|
2077
|
+
for output_path in output_paths["json"]:
|
|
2078
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
2079
|
+
output_path.write_text(
|
|
2080
|
+
json.dumps(payload, indent=2, sort_keys=True, default=str),
|
|
2081
|
+
encoding="utf-8",
|
|
2082
|
+
)
|
|
2083
|
+
written.append(str(output_path))
|
|
2084
|
+
for output_path in output_paths["junit"]:
|
|
2085
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
2086
|
+
output_path.write_text(render_junit_fn(payload), encoding="utf-8")
|
|
2087
|
+
written.append(str(output_path))
|
|
2088
|
+
for output_path in output_paths["sarif"]:
|
|
2089
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
2090
|
+
output_path.write_text(
|
|
2091
|
+
render_sarif_fn(payload, manifest_path=path),
|
|
2092
|
+
encoding="utf-8",
|
|
2093
|
+
)
|
|
2094
|
+
written.append(str(output_path))
|
|
2095
|
+
for output_path in output_paths["markdown"]:
|
|
2096
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
2097
|
+
output_path.write_text(
|
|
2098
|
+
render_markdown_fn(payload, source_path=path),
|
|
2099
|
+
encoding="utf-8",
|
|
2100
|
+
)
|
|
2101
|
+
written.append(str(output_path))
|
|
2102
|
+
return written
|
|
2103
|
+
|
|
2104
|
+
|
|
2105
|
+
def _child_renderers(command: str) -> tuple[Any, Any, Any]:
|
|
2106
|
+
if command == "suite":
|
|
2107
|
+
return render_junit, render_sarif, render_markdown
|
|
2108
|
+
if command == "action_run":
|
|
2109
|
+
from fi.alk import actions, simulate
|
|
2110
|
+
|
|
2111
|
+
def render_action_run_markdown(
|
|
2112
|
+
payload: Mapping[str, Any],
|
|
2113
|
+
*,
|
|
2114
|
+
source_path: Path,
|
|
2115
|
+
) -> str:
|
|
2116
|
+
return actions.render_action_run_markdown(payload)
|
|
2117
|
+
|
|
2118
|
+
return simulate.render_junit, simulate.render_sarif, render_action_run_markdown
|
|
2119
|
+
if command == "redteam":
|
|
2120
|
+
from fi.alk import redteam
|
|
2121
|
+
|
|
2122
|
+
return redteam.render_junit, redteam.render_sarif, redteam.render_markdown
|
|
2123
|
+
from fi.alk import simulate
|
|
2124
|
+
|
|
2125
|
+
return simulate.render_junit, simulate.render_sarif, simulate.render_markdown
|
|
2126
|
+
|
|
2127
|
+
|
|
2128
|
+
def _suite_child_evidence(
|
|
2129
|
+
job: Mapping[str, Any],
|
|
2130
|
+
child: Mapping[str, Any],
|
|
2131
|
+
*,
|
|
2132
|
+
base_dir: Path,
|
|
2133
|
+
) -> dict[str, Any]:
|
|
2134
|
+
role = _suite_evidence_role(job, child)
|
|
2135
|
+
exit_code = int(child.get("exit_code", 1))
|
|
2136
|
+
manifest_path = Path(str(child.get("path") or ""))
|
|
2137
|
+
replay_class = str(
|
|
2138
|
+
job.get("replay_class")
|
|
2139
|
+
or job.get("replay")
|
|
2140
|
+
or _as_mapping(job.get("metadata")).get("replay_class")
|
|
2141
|
+
or "r0"
|
|
2142
|
+
)
|
|
2143
|
+
output_digests = _suite_output_digests(
|
|
2144
|
+
child.get("outputs_written"),
|
|
2145
|
+
base_dir=base_dir,
|
|
2146
|
+
)
|
|
2147
|
+
freeze = {
|
|
2148
|
+
"kind": "agent-learning.suite.evidence-freeze.v1",
|
|
2149
|
+
"hash_algorithm": "sha256",
|
|
2150
|
+
"replay_class": replay_class,
|
|
2151
|
+
"manifest": _suite_file_digest(manifest_path),
|
|
2152
|
+
"result_sha256": _suite_json_digest(child.get("result")),
|
|
2153
|
+
"outputs": output_digests,
|
|
2154
|
+
"outputs_sha256": _suite_json_digest(output_digests),
|
|
2155
|
+
}
|
|
2156
|
+
freeze["content_addressed"] = bool(
|
|
2157
|
+
_as_mapping(freeze.get("manifest")).get("sha256")
|
|
2158
|
+
and freeze.get("result_sha256")
|
|
2159
|
+
)
|
|
2160
|
+
reasons: list[str] = []
|
|
2161
|
+
if exit_code != 0:
|
|
2162
|
+
status = "rejected"
|
|
2163
|
+
admitted = False
|
|
2164
|
+
reasons.append("child_failed")
|
|
2165
|
+
elif role in _ADMITTED_EVIDENCE_ROLES:
|
|
2166
|
+
status = "admitted"
|
|
2167
|
+
admitted = True
|
|
2168
|
+
else:
|
|
2169
|
+
status = role if role in _NON_ADMITTED_EVIDENCE_ROLES else "diagnostic"
|
|
2170
|
+
admitted = False
|
|
2171
|
+
reasons.append(f"evidence_role_{status}")
|
|
2172
|
+
if _suite_path_is_fixture(child.get("path")) and status != "rejected":
|
|
2173
|
+
role = "fixture"
|
|
2174
|
+
status = "fixture"
|
|
2175
|
+
admitted = False
|
|
2176
|
+
if "fixture_path" not in reasons:
|
|
2177
|
+
reasons.append("fixture_path")
|
|
2178
|
+
metadata = _as_mapping(job.get("metadata"))
|
|
2179
|
+
claim_scope = (
|
|
2180
|
+
job.get("claim_scope")
|
|
2181
|
+
or job.get("claim")
|
|
2182
|
+
or metadata.get("claim_scope")
|
|
2183
|
+
or ("paper_facing" if admitted else "audit")
|
|
2184
|
+
)
|
|
2185
|
+
return {
|
|
2186
|
+
"kind": "agent-learning.suite.evidence-row.v1",
|
|
2187
|
+
"row_id": str(child.get("id") or job.get("id") or ""),
|
|
2188
|
+
"status": status,
|
|
2189
|
+
"role": role,
|
|
2190
|
+
"admitted": admitted,
|
|
2191
|
+
"reason": reasons,
|
|
2192
|
+
"claim_scope": str(claim_scope),
|
|
2193
|
+
"workload": str(job.get("workload") or job.get("id") or child.get("id") or ""),
|
|
2194
|
+
"driver": str(job.get("driver") or child.get("command") or ""),
|
|
2195
|
+
"command": child.get("command"),
|
|
2196
|
+
"path": child.get("path"),
|
|
2197
|
+
"result_kind": child.get("kind"),
|
|
2198
|
+
"exit_code": exit_code,
|
|
2199
|
+
"provenance": {
|
|
2200
|
+
"job_id": child.get("id") or job.get("id"),
|
|
2201
|
+
"job_name": child.get("name") or job.get("name"),
|
|
2202
|
+
"manifest_path": child.get("path"),
|
|
2203
|
+
"manifest_sha256": _as_mapping(freeze["manifest"]).get("sha256"),
|
|
2204
|
+
"result_sha256": freeze.get("result_sha256"),
|
|
2205
|
+
"outputs_written": list(child.get("outputs_written") or []),
|
|
2206
|
+
"output_digests": output_digests,
|
|
2207
|
+
"outputs_sha256": freeze.get("outputs_sha256"),
|
|
2208
|
+
"replay_class": replay_class,
|
|
2209
|
+
"content_addressed": freeze["content_addressed"],
|
|
2210
|
+
},
|
|
2211
|
+
"freeze": freeze,
|
|
2212
|
+
}
|
|
2213
|
+
|
|
2214
|
+
|
|
2215
|
+
def _suite_evidence_role(
|
|
2216
|
+
job: Mapping[str, Any],
|
|
2217
|
+
child: Mapping[str, Any],
|
|
2218
|
+
) -> str:
|
|
2219
|
+
metadata = _as_mapping(job.get("metadata"))
|
|
2220
|
+
raw = (
|
|
2221
|
+
job.get("evidence_role")
|
|
2222
|
+
or job.get("evidence_status")
|
|
2223
|
+
or job.get("evidence")
|
|
2224
|
+
or metadata.get("evidence_role")
|
|
2225
|
+
or metadata.get("evidence_status")
|
|
2226
|
+
)
|
|
2227
|
+
role = _suite_key(raw) if raw is not None else ""
|
|
2228
|
+
if role in _ADMITTED_EVIDENCE_ROLES or role in _NON_ADMITTED_EVIDENCE_ROLES:
|
|
2229
|
+
return role
|
|
2230
|
+
if _suite_path_is_fixture(child.get("path") or job.get("path")):
|
|
2231
|
+
return "fixture"
|
|
2232
|
+
return "admitted"
|
|
2233
|
+
|
|
2234
|
+
|
|
2235
|
+
def _suite_path_is_fixture(value: Any) -> bool:
|
|
2236
|
+
text = str(value or "").replace("\\", "/").lower()
|
|
2237
|
+
return "/fixtures/" in text or text.startswith("fixtures/")
|
|
2238
|
+
|
|
2239
|
+
|
|
2240
|
+
def _suite_file_digest(path: str | Path) -> dict[str, Any]:
|
|
2241
|
+
file_path = Path(path).expanduser()
|
|
2242
|
+
exists = file_path.exists()
|
|
2243
|
+
if not exists or not file_path.is_file():
|
|
2244
|
+
return {
|
|
2245
|
+
"path": str(file_path),
|
|
2246
|
+
"exists": exists,
|
|
2247
|
+
"sha256": None,
|
|
2248
|
+
"bytes": 0,
|
|
2249
|
+
}
|
|
2250
|
+
data = file_path.read_bytes()
|
|
2251
|
+
return {
|
|
2252
|
+
"path": str(file_path),
|
|
2253
|
+
"exists": True,
|
|
2254
|
+
"sha256": hashlib.sha256(data).hexdigest(),
|
|
2255
|
+
"bytes": len(data),
|
|
2256
|
+
}
|
|
2257
|
+
|
|
2258
|
+
|
|
2259
|
+
def _suite_json_digest(value: Any) -> str:
|
|
2260
|
+
data = json.dumps(
|
|
2261
|
+
value,
|
|
2262
|
+
sort_keys=True,
|
|
2263
|
+
separators=(",", ":"),
|
|
2264
|
+
default=str,
|
|
2265
|
+
).encode("utf-8")
|
|
2266
|
+
return hashlib.sha256(data).hexdigest()
|
|
2267
|
+
|
|
2268
|
+
|
|
2269
|
+
def _suite_output_digests(
|
|
2270
|
+
values: Any,
|
|
2271
|
+
*,
|
|
2272
|
+
base_dir: Path,
|
|
2273
|
+
) -> list[dict[str, Any]]:
|
|
2274
|
+
records: list[dict[str, Any]] = []
|
|
2275
|
+
for value in _as_list(values):
|
|
2276
|
+
path = Path(str(value)).expanduser()
|
|
2277
|
+
if not path.is_absolute():
|
|
2278
|
+
path = (base_dir / path).resolve()
|
|
2279
|
+
records.append(_suite_file_digest(path))
|
|
2280
|
+
return records
|
|
2281
|
+
|
|
2282
|
+
|
|
2283
|
+
def _suite_evidence_admission(
|
|
2284
|
+
children: Sequence[Mapping[str, Any]],
|
|
2285
|
+
) -> dict[str, Any]:
|
|
2286
|
+
rows = [copy.deepcopy(dict(_as_mapping(child.get("evidence")))) for child in children]
|
|
2287
|
+
rows = [row for row in rows if row]
|
|
2288
|
+
by_status: dict[str, int] = {}
|
|
2289
|
+
by_role: dict[str, int] = {}
|
|
2290
|
+
for row in rows:
|
|
2291
|
+
status = str(row.get("status") or "unknown")
|
|
2292
|
+
role = str(row.get("role") or "unknown")
|
|
2293
|
+
by_status[status] = by_status.get(status, 0) + 1
|
|
2294
|
+
by_role[role] = by_role.get(role, 0) + 1
|
|
2295
|
+
admitted_rows = [row for row in rows if bool(row.get("admitted"))]
|
|
2296
|
+
rejected_rows = [row for row in rows if str(row.get("status") or "") == "rejected"]
|
|
2297
|
+
non_admitted_rows = [row for row in rows if not bool(row.get("admitted"))]
|
|
2298
|
+
frozen_rows = [row for row in rows if _suite_row_content_addressed(row)]
|
|
2299
|
+
admitted_unfrozen_rows = [
|
|
2300
|
+
row for row in admitted_rows if not _suite_row_content_addressed(row)
|
|
2301
|
+
]
|
|
2302
|
+
return {
|
|
2303
|
+
"kind": "agent-learning.suite.evidence-admission.v1",
|
|
2304
|
+
"admitted_count": len(admitted_rows),
|
|
2305
|
+
"non_admitted_count": len(non_admitted_rows),
|
|
2306
|
+
"rejected_count": len(rejected_rows),
|
|
2307
|
+
"frozen_count": len(frozen_rows),
|
|
2308
|
+
"unfrozen_count": len(rows) - len(frozen_rows),
|
|
2309
|
+
"admitted_frozen_count": len(admitted_rows) - len(admitted_unfrozen_rows),
|
|
2310
|
+
"by_status": dict(sorted(by_status.items())),
|
|
2311
|
+
"by_role": dict(sorted(by_role.items())),
|
|
2312
|
+
"admitted_row_ids": [str(row.get("row_id")) for row in admitted_rows],
|
|
2313
|
+
"non_admitted_row_ids": [str(row.get("row_id")) for row in non_admitted_rows],
|
|
2314
|
+
"admitted_unfrozen_row_ids": [
|
|
2315
|
+
str(row.get("row_id")) for row in admitted_unfrozen_rows
|
|
2316
|
+
],
|
|
2317
|
+
"rows": rows,
|
|
2318
|
+
}
|
|
2319
|
+
|
|
2320
|
+
|
|
2321
|
+
def _suite_row_content_addressed(row: Mapping[str, Any]) -> bool:
|
|
2322
|
+
freeze = _as_mapping(row.get("freeze"))
|
|
2323
|
+
return bool(freeze.get("content_addressed"))
|
|
2324
|
+
|
|
2325
|
+
|
|
2326
|
+
def _suite_evidence_policy(suite: Mapping[str, Any]) -> dict[str, Any]:
|
|
2327
|
+
raw = (
|
|
2328
|
+
suite.get("evidence_policy")
|
|
2329
|
+
or suite.get("evidence_admission_policy")
|
|
2330
|
+
or suite.get("admission_policy")
|
|
2331
|
+
or {}
|
|
2332
|
+
)
|
|
2333
|
+
if isinstance(raw, Mapping):
|
|
2334
|
+
policy = copy.deepcopy(dict(raw))
|
|
2335
|
+
else:
|
|
2336
|
+
policy = {}
|
|
2337
|
+
min_admitted = policy.get("min_admitted")
|
|
2338
|
+
if min_admitted is None and bool(policy.get("require_admitted")):
|
|
2339
|
+
min_admitted = 1
|
|
2340
|
+
policy["min_admitted"] = int(min_admitted or 0)
|
|
2341
|
+
policy["require_freeze"] = bool(
|
|
2342
|
+
policy.get("require_freeze") or policy.get("require_content_addressed")
|
|
2343
|
+
)
|
|
2344
|
+
return policy
|
|
2345
|
+
|
|
2346
|
+
|
|
2347
|
+
def _suite_optimizer_governance_policy(suite: Mapping[str, Any]) -> dict[str, Any]:
|
|
2348
|
+
raw = (
|
|
2349
|
+
suite.get("optimizer_governance_policy")
|
|
2350
|
+
or suite.get("optimization_governance_policy")
|
|
2351
|
+
or {}
|
|
2352
|
+
)
|
|
2353
|
+
policy = copy.deepcopy(dict(raw)) if isinstance(raw, Mapping) else {}
|
|
2354
|
+
required = bool(
|
|
2355
|
+
policy.get("require_optimizer_governance")
|
|
2356
|
+
or policy.get("required")
|
|
2357
|
+
or policy.get("require_passed")
|
|
2358
|
+
)
|
|
2359
|
+
min_governed = policy.get("min_governed")
|
|
2360
|
+
if min_governed is None and required:
|
|
2361
|
+
min_governed = 1
|
|
2362
|
+
commands = _unique_strings(
|
|
2363
|
+
policy.get("commands")
|
|
2364
|
+
or policy.get("target_commands")
|
|
2365
|
+
or ["optimize"]
|
|
2366
|
+
)
|
|
2367
|
+
policy["require_optimizer_governance"] = required
|
|
2368
|
+
policy["require_passed"] = bool(policy.get("require_passed") or required)
|
|
2369
|
+
policy["fail_on_warning"] = bool(policy.get("fail_on_warning"))
|
|
2370
|
+
policy["min_governed"] = int(min_governed or 0)
|
|
2371
|
+
policy["commands"] = commands or ["optimize"]
|
|
2372
|
+
return policy
|
|
2373
|
+
|
|
2374
|
+
|
|
2375
|
+
def _suite_optimizer_governance(
|
|
2376
|
+
children: Sequence[Mapping[str, Any]],
|
|
2377
|
+
policy: Mapping[str, Any],
|
|
2378
|
+
) -> dict[str, Any]:
|
|
2379
|
+
target_commands = {
|
|
2380
|
+
_normalize_command(command)
|
|
2381
|
+
for command in _as_list(policy.get("commands"))
|
|
2382
|
+
if command
|
|
2383
|
+
}
|
|
2384
|
+
rows = [
|
|
2385
|
+
_suite_optimizer_governance_row(child)
|
|
2386
|
+
for child in children
|
|
2387
|
+
if _suite_optimizer_governance_targets_child(child, target_commands)
|
|
2388
|
+
]
|
|
2389
|
+
governed_rows = [row for row in rows if bool(row.get("governance_present"))]
|
|
2390
|
+
failed_rows = [
|
|
2391
|
+
row
|
|
2392
|
+
for row in governed_rows
|
|
2393
|
+
if row.get("governance_status") != "passed" or row.get("passed") is False
|
|
2394
|
+
]
|
|
2395
|
+
missing_rows = [row for row in rows if not bool(row.get("governance_present"))]
|
|
2396
|
+
warning_rows = [
|
|
2397
|
+
row
|
|
2398
|
+
for row in governed_rows
|
|
2399
|
+
if _as_list(row.get("warning_check_ids"))
|
|
2400
|
+
]
|
|
2401
|
+
return {
|
|
2402
|
+
"kind": "agent-learning.suite.optimizer-governance.v1",
|
|
2403
|
+
"status": "failed" if failed_rows or missing_rows else "passed",
|
|
2404
|
+
"policy": copy.deepcopy(dict(policy)),
|
|
2405
|
+
"target_count": len(rows),
|
|
2406
|
+
"governed_count": len(governed_rows),
|
|
2407
|
+
"passed_count": len(governed_rows) - len(failed_rows),
|
|
2408
|
+
"failed_count": len(failed_rows),
|
|
2409
|
+
"missing_count": len(missing_rows),
|
|
2410
|
+
"warning_count": len(warning_rows),
|
|
2411
|
+
"target_child_ids": [str(row.get("child_id")) for row in rows],
|
|
2412
|
+
"governed_child_ids": [str(row.get("child_id")) for row in governed_rows],
|
|
2413
|
+
"failed_child_ids": [str(row.get("child_id")) for row in failed_rows],
|
|
2414
|
+
"missing_child_ids": [str(row.get("child_id")) for row in missing_rows],
|
|
2415
|
+
"warning_child_ids": [str(row.get("child_id")) for row in warning_rows],
|
|
2416
|
+
"rows": rows,
|
|
2417
|
+
}
|
|
2418
|
+
|
|
2419
|
+
|
|
2420
|
+
def _suite_optimizer_governance_targets_child(
|
|
2421
|
+
child: Mapping[str, Any],
|
|
2422
|
+
target_commands: set[str],
|
|
2423
|
+
) -> bool:
|
|
2424
|
+
result = _as_mapping(child.get("result"))
|
|
2425
|
+
if _as_mapping(result.get("optimization_governance")):
|
|
2426
|
+
return True
|
|
2427
|
+
command = _normalize_command(child.get("command") or "")
|
|
2428
|
+
if command in target_commands:
|
|
2429
|
+
return True
|
|
2430
|
+
return False
|
|
2431
|
+
|
|
2432
|
+
|
|
2433
|
+
def _suite_optimizer_governance_row(child: Mapping[str, Any]) -> dict[str, Any]:
|
|
2434
|
+
result = _as_mapping(child.get("result"))
|
|
2435
|
+
governance = _as_mapping(result.get("optimization_governance"))
|
|
2436
|
+
if not governance:
|
|
2437
|
+
governance = _as_mapping(_as_mapping(result.get("optimization")).get("governance"))
|
|
2438
|
+
evidence = _as_mapping(governance.get("evidence"))
|
|
2439
|
+
return {
|
|
2440
|
+
"kind": "agent-learning.suite.optimizer-governance-row.v1",
|
|
2441
|
+
"child_id": child.get("id"),
|
|
2442
|
+
"command": child.get("command"),
|
|
2443
|
+
"path": child.get("path"),
|
|
2444
|
+
"result_kind": child.get("kind"),
|
|
2445
|
+
"child_status": child.get("status"),
|
|
2446
|
+
"child_exit_code": int(child.get("exit_code", 1)),
|
|
2447
|
+
"governance_present": bool(governance),
|
|
2448
|
+
"governance_kind": governance.get("kind"),
|
|
2449
|
+
"governance_status": governance.get("status") if governance else "missing",
|
|
2450
|
+
"passed": bool(governance.get("passed")) if governance else False,
|
|
2451
|
+
"selected_candidate_id": governance.get("selected_candidate_id"),
|
|
2452
|
+
"selected_rank": governance.get("selected_rank"),
|
|
2453
|
+
"check_count": int(governance.get("check_count") or 0),
|
|
2454
|
+
"failed_check_ids": [
|
|
2455
|
+
str(item) for item in _as_list(governance.get("failed_check_ids"))
|
|
2456
|
+
],
|
|
2457
|
+
"warning_check_ids": [
|
|
2458
|
+
str(item) for item in _as_list(governance.get("warning_check_ids"))
|
|
2459
|
+
],
|
|
2460
|
+
"candidate_count": int(evidence.get("candidate_count") or 0),
|
|
2461
|
+
"content_addressed_count": int(
|
|
2462
|
+
evidence.get("content_addressed_count") or 0
|
|
2463
|
+
),
|
|
2464
|
+
"metric_count": int(evidence.get("metric_count") or 0),
|
|
2465
|
+
"patch_path_count": int(evidence.get("patch_path_count") or 0),
|
|
2466
|
+
}
|
|
2467
|
+
|
|
2468
|
+
|
|
2469
|
+
def _suite_optimizer_governance_findings(
|
|
2470
|
+
optimizer_governance: Mapping[str, Any],
|
|
2471
|
+
policy: Mapping[str, Any],
|
|
2472
|
+
) -> list[dict[str, Any]]:
|
|
2473
|
+
findings: list[dict[str, Any]] = []
|
|
2474
|
+
min_governed = int(policy.get("min_governed") or 0)
|
|
2475
|
+
governed_count = int(optimizer_governance.get("governed_count") or 0)
|
|
2476
|
+
if min_governed > governed_count:
|
|
2477
|
+
findings.append({
|
|
2478
|
+
"type": "suite_optimizer_governance_missing",
|
|
2479
|
+
"level": "error",
|
|
2480
|
+
"reason": (
|
|
2481
|
+
f"Suite optimizer governance gate requires at least {min_governed} "
|
|
2482
|
+
f"governed optimizer child row(s), but only {governed_count} "
|
|
2483
|
+
"were found."
|
|
2484
|
+
),
|
|
2485
|
+
"min_governed": min_governed,
|
|
2486
|
+
"governed_count": governed_count,
|
|
2487
|
+
"missing_child_ids": list(
|
|
2488
|
+
optimizer_governance.get("missing_child_ids") or []
|
|
2489
|
+
),
|
|
2490
|
+
})
|
|
2491
|
+
if bool(policy.get("require_passed")):
|
|
2492
|
+
failed_child_ids = list(optimizer_governance.get("failed_child_ids") or [])
|
|
2493
|
+
missing_child_ids = list(optimizer_governance.get("missing_child_ids") or [])
|
|
2494
|
+
blocked_child_ids = sorted(
|
|
2495
|
+
{str(item) for item in [*failed_child_ids, *missing_child_ids]}
|
|
2496
|
+
)
|
|
2497
|
+
if blocked_child_ids:
|
|
2498
|
+
findings.append({
|
|
2499
|
+
"type": "suite_optimizer_governance_failed",
|
|
2500
|
+
"level": "error",
|
|
2501
|
+
"reason": (
|
|
2502
|
+
"Suite optimizer governance gate requires passed governance "
|
|
2503
|
+
f"for optimizer children, but {len(blocked_child_ids)} child "
|
|
2504
|
+
"row(s) are missing or failed."
|
|
2505
|
+
),
|
|
2506
|
+
"failed_child_ids": failed_child_ids,
|
|
2507
|
+
"missing_child_ids": missing_child_ids,
|
|
2508
|
+
})
|
|
2509
|
+
if bool(policy.get("fail_on_warning")):
|
|
2510
|
+
warning_child_ids = list(optimizer_governance.get("warning_child_ids") or [])
|
|
2511
|
+
if warning_child_ids:
|
|
2512
|
+
findings.append({
|
|
2513
|
+
"type": "suite_optimizer_governance_warning",
|
|
2514
|
+
"level": "error",
|
|
2515
|
+
"reason": (
|
|
2516
|
+
"Suite optimizer governance gate is configured to fail on "
|
|
2517
|
+
f"warnings, and {len(warning_child_ids)} child row(s) have "
|
|
2518
|
+
"governance warnings."
|
|
2519
|
+
),
|
|
2520
|
+
"warning_child_ids": warning_child_ids,
|
|
2521
|
+
})
|
|
2522
|
+
return findings
|
|
2523
|
+
|
|
2524
|
+
|
|
2525
|
+
def _suite_evidence_findings(
|
|
2526
|
+
admission: Mapping[str, Any],
|
|
2527
|
+
policy: Mapping[str, Any],
|
|
2528
|
+
) -> list[dict[str, Any]]:
|
|
2529
|
+
min_admitted = int(policy.get("min_admitted") or 0)
|
|
2530
|
+
admitted_count = int(admission.get("admitted_count") or 0)
|
|
2531
|
+
findings: list[dict[str, Any]] = []
|
|
2532
|
+
if min_admitted > admitted_count:
|
|
2533
|
+
findings.append({
|
|
2534
|
+
"type": "suite_evidence_admission_missing",
|
|
2535
|
+
"level": "error",
|
|
2536
|
+
"reason": (
|
|
2537
|
+
f"Suite evidence gate requires at least {min_admitted} admitted "
|
|
2538
|
+
f"row(s), but only {admitted_count} were admitted."
|
|
2539
|
+
),
|
|
2540
|
+
"admitted_count": admitted_count,
|
|
2541
|
+
"min_admitted": min_admitted,
|
|
2542
|
+
})
|
|
2543
|
+
if bool(policy.get("require_freeze")):
|
|
2544
|
+
missing = [
|
|
2545
|
+
str(row_id)
|
|
2546
|
+
for row_id in _as_list(admission.get("admitted_unfrozen_row_ids"))
|
|
2547
|
+
]
|
|
2548
|
+
if missing:
|
|
2549
|
+
findings.append({
|
|
2550
|
+
"type": "suite_evidence_freeze_missing",
|
|
2551
|
+
"level": "error",
|
|
2552
|
+
"reason": (
|
|
2553
|
+
"Suite evidence gate requires content-addressed admitted "
|
|
2554
|
+
f"rows, but {len(missing)} admitted row(s) are missing "
|
|
2555
|
+
"manifest/result digests."
|
|
2556
|
+
),
|
|
2557
|
+
"missing": missing,
|
|
2558
|
+
})
|
|
2559
|
+
return findings
|
|
2560
|
+
|
|
2561
|
+
|
|
2562
|
+
def _suite_result(
|
|
2563
|
+
*,
|
|
2564
|
+
suite: Mapping[str, Any],
|
|
2565
|
+
suite_path: Path,
|
|
2566
|
+
children: Sequence[Mapping[str, Any]],
|
|
2567
|
+
name: Optional[str],
|
|
2568
|
+
dry_run: bool,
|
|
2569
|
+
fail_fast: bool,
|
|
2570
|
+
duration_seconds: float,
|
|
2571
|
+
) -> dict[str, Any]:
|
|
2572
|
+
job_count = len(_suite_jobs(suite))
|
|
2573
|
+
passed = [child for child in children if int(child.get("exit_code", 1)) == 0]
|
|
2574
|
+
failed = [child for child in children if int(child.get("exit_code", 1)) != 0]
|
|
2575
|
+
score = round(len(passed) / job_count, 4) if job_count else 0.0
|
|
2576
|
+
command_counts: dict[str, int] = {}
|
|
2577
|
+
for child in children:
|
|
2578
|
+
command = str(child.get("command") or "unknown")
|
|
2579
|
+
command_counts[command] = command_counts.get(command, 0) + 1
|
|
2580
|
+
capabilities = _suite_capability_summary(children)
|
|
2581
|
+
required_capabilities = _suite_required_capabilities(suite)
|
|
2582
|
+
missing_capabilities = _missing_required_capabilities(
|
|
2583
|
+
required_capabilities,
|
|
2584
|
+
capabilities,
|
|
2585
|
+
)
|
|
2586
|
+
capability_findings = _suite_capability_findings(missing_capabilities)
|
|
2587
|
+
framework_coverage = _suite_framework_coverage(
|
|
2588
|
+
children,
|
|
2589
|
+
required_frameworks=required_capabilities.get("frameworks", []),
|
|
2590
|
+
)
|
|
2591
|
+
framework_findings = _suite_framework_findings(framework_coverage)
|
|
2592
|
+
evidence_admission = _suite_evidence_admission(children)
|
|
2593
|
+
evidence_policy = _suite_evidence_policy(suite)
|
|
2594
|
+
evidence_findings = _suite_evidence_findings(
|
|
2595
|
+
evidence_admission,
|
|
2596
|
+
evidence_policy,
|
|
2597
|
+
)
|
|
2598
|
+
optimizer_governance_policy = _suite_optimizer_governance_policy(suite)
|
|
2599
|
+
optimizer_governance = _suite_optimizer_governance(
|
|
2600
|
+
children,
|
|
2601
|
+
optimizer_governance_policy,
|
|
2602
|
+
)
|
|
2603
|
+
optimizer_governance_findings = _suite_optimizer_governance_findings(
|
|
2604
|
+
optimizer_governance,
|
|
2605
|
+
optimizer_governance_policy,
|
|
2606
|
+
)
|
|
2607
|
+
suite_findings = [
|
|
2608
|
+
*capability_findings,
|
|
2609
|
+
*framework_findings,
|
|
2610
|
+
*evidence_findings,
|
|
2611
|
+
*optimizer_governance_findings,
|
|
2612
|
+
*_suite_findings(children),
|
|
2613
|
+
]
|
|
2614
|
+
suite_passed = (
|
|
2615
|
+
len(failed) == 0
|
|
2616
|
+
and len(children) == job_count
|
|
2617
|
+
and not capability_findings
|
|
2618
|
+
and not framework_findings
|
|
2619
|
+
and not evidence_findings
|
|
2620
|
+
and not optimizer_governance_findings
|
|
2621
|
+
)
|
|
2622
|
+
trust_certificate = _suite_trust_certificate(
|
|
2623
|
+
suite=suite,
|
|
2624
|
+
suite_path=suite_path,
|
|
2625
|
+
children=children,
|
|
2626
|
+
capabilities=capabilities,
|
|
2627
|
+
framework_coverage=framework_coverage,
|
|
2628
|
+
evidence_admission=evidence_admission,
|
|
2629
|
+
optimizer_governance=optimizer_governance,
|
|
2630
|
+
missing_capabilities=missing_capabilities,
|
|
2631
|
+
suite_passed=suite_passed,
|
|
2632
|
+
job_count=job_count,
|
|
2633
|
+
executed_count=len(children),
|
|
2634
|
+
passed_count=len(passed),
|
|
2635
|
+
failed_count=len(failed),
|
|
2636
|
+
score=score,
|
|
2637
|
+
)
|
|
2638
|
+
return {
|
|
2639
|
+
"kind": AGENT_LEARNING_SUITE_KIND,
|
|
2640
|
+
"version": AGENT_LEARNING_SUITE_KIND,
|
|
2641
|
+
"name": str(name or suite.get("name") or suite_path.stem),
|
|
2642
|
+
"status": "passed" if suite_passed else "failed",
|
|
2643
|
+
"exit_code": 0 if suite_passed else 1,
|
|
2644
|
+
"dry_run": dry_run,
|
|
2645
|
+
"fail_fast": fail_fast,
|
|
2646
|
+
"summary": {
|
|
2647
|
+
"job_count": job_count,
|
|
2648
|
+
"executed_count": len(children),
|
|
2649
|
+
"passed_count": len(passed),
|
|
2650
|
+
"failed_count": len(failed),
|
|
2651
|
+
"skipped_count": max(job_count - len(children), 0),
|
|
2652
|
+
"score": score,
|
|
2653
|
+
"trust_certificate_verdict": trust_certificate["verdict"],
|
|
2654
|
+
"trust_certificate_assurance_level": trust_certificate[
|
|
2655
|
+
"assurance_level"
|
|
2656
|
+
],
|
|
2657
|
+
"trust_certificate_promotion_ready": trust_certificate[
|
|
2658
|
+
"promotion_ready"
|
|
2659
|
+
],
|
|
2660
|
+
"trust_certificate_failed_gate_count": len(
|
|
2661
|
+
trust_certificate["failed_gate_ids"]
|
|
2662
|
+
),
|
|
2663
|
+
"trust_certificate_conditional_gate_count": len(
|
|
2664
|
+
trust_certificate["conditional_gate_ids"]
|
|
2665
|
+
),
|
|
2666
|
+
"commands": command_counts,
|
|
2667
|
+
"capabilities": capabilities,
|
|
2668
|
+
"required_capabilities": required_capabilities,
|
|
2669
|
+
"missing_required_capabilities": missing_capabilities,
|
|
2670
|
+
"capability_gate_passed": not capability_findings,
|
|
2671
|
+
"framework_coverage_passed": not framework_findings,
|
|
2672
|
+
"observed_framework_count": framework_coverage["observed_count"],
|
|
2673
|
+
"required_framework_count": framework_coverage["required_count"],
|
|
2674
|
+
"missing_framework_count": framework_coverage["missing_count"],
|
|
2675
|
+
"adapter_conformance_failed_count": framework_coverage[
|
|
2676
|
+
"adapter_conformance_failed_count"
|
|
2677
|
+
],
|
|
2678
|
+
"framework_coverage": {
|
|
2679
|
+
key: value
|
|
2680
|
+
for key, value in framework_coverage.items()
|
|
2681
|
+
if key != "rows"
|
|
2682
|
+
},
|
|
2683
|
+
"evidence_gate_passed": not evidence_findings,
|
|
2684
|
+
"optimizer_governance_gate_passed": not optimizer_governance_findings,
|
|
2685
|
+
"optimizer_governance_policy": optimizer_governance_policy,
|
|
2686
|
+
"optimizer_governance_target_count": optimizer_governance[
|
|
2687
|
+
"target_count"
|
|
2688
|
+
],
|
|
2689
|
+
"optimizer_governance_governed_count": optimizer_governance[
|
|
2690
|
+
"governed_count"
|
|
2691
|
+
],
|
|
2692
|
+
"optimizer_governance_passed_count": optimizer_governance[
|
|
2693
|
+
"passed_count"
|
|
2694
|
+
],
|
|
2695
|
+
"optimizer_governance_failed_count": optimizer_governance[
|
|
2696
|
+
"failed_count"
|
|
2697
|
+
],
|
|
2698
|
+
"optimizer_governance_missing_count": optimizer_governance[
|
|
2699
|
+
"missing_count"
|
|
2700
|
+
],
|
|
2701
|
+
"optimizer_governance_warning_count": optimizer_governance[
|
|
2702
|
+
"warning_count"
|
|
2703
|
+
],
|
|
2704
|
+
"admitted_evidence_count": evidence_admission["admitted_count"],
|
|
2705
|
+
"non_admitted_evidence_count": evidence_admission[
|
|
2706
|
+
"non_admitted_count"
|
|
2707
|
+
],
|
|
2708
|
+
"rejected_evidence_count": evidence_admission["rejected_count"],
|
|
2709
|
+
"frozen_evidence_count": evidence_admission["frozen_count"],
|
|
2710
|
+
"unfrozen_evidence_count": evidence_admission["unfrozen_count"],
|
|
2711
|
+
"admitted_frozen_evidence_count": evidence_admission[
|
|
2712
|
+
"admitted_frozen_count"
|
|
2713
|
+
],
|
|
2714
|
+
"evidence_admission": {
|
|
2715
|
+
key: value
|
|
2716
|
+
for key, value in evidence_admission.items()
|
|
2717
|
+
if key != "rows"
|
|
2718
|
+
},
|
|
2719
|
+
},
|
|
2720
|
+
"framework_coverage": framework_coverage,
|
|
2721
|
+
"evidence_admission": evidence_admission,
|
|
2722
|
+
"optimizer_governance": optimizer_governance,
|
|
2723
|
+
"trust_certificate": trust_certificate,
|
|
2724
|
+
"children": list(children),
|
|
2725
|
+
"jobs": list(children),
|
|
2726
|
+
"findings": suite_findings,
|
|
2727
|
+
"duration_seconds": duration_seconds,
|
|
2728
|
+
}
|
|
2729
|
+
|
|
2730
|
+
|
|
2731
|
+
def _suite_descriptor(suite: Mapping[str, Any]) -> dict[str, Any]:
|
|
2732
|
+
return {
|
|
2733
|
+
"version": suite.get("version") or AGENT_LEARNING_SUITE_KIND,
|
|
2734
|
+
"name": suite.get("name"),
|
|
2735
|
+
"job_count": len(_suite_jobs(suite)),
|
|
2736
|
+
"jobs": [
|
|
2737
|
+
{
|
|
2738
|
+
"id": job.get("id"),
|
|
2739
|
+
"command": job.get("command"),
|
|
2740
|
+
"path": job.get("path"),
|
|
2741
|
+
}
|
|
2742
|
+
for job in _suite_jobs(suite)
|
|
2743
|
+
],
|
|
2744
|
+
"required_capabilities": _suite_required_capabilities(suite),
|
|
2745
|
+
}
|
|
2746
|
+
|
|
2747
|
+
|
|
2748
|
+
def _suite_trust_certificate(
|
|
2749
|
+
*,
|
|
2750
|
+
suite: Mapping[str, Any],
|
|
2751
|
+
suite_path: Path,
|
|
2752
|
+
children: Sequence[Mapping[str, Any]],
|
|
2753
|
+
capabilities: Mapping[str, Sequence[str]],
|
|
2754
|
+
framework_coverage: Mapping[str, Any],
|
|
2755
|
+
evidence_admission: Mapping[str, Any],
|
|
2756
|
+
optimizer_governance: Mapping[str, Any],
|
|
2757
|
+
missing_capabilities: Mapping[str, Sequence[str]],
|
|
2758
|
+
suite_passed: bool,
|
|
2759
|
+
job_count: int,
|
|
2760
|
+
executed_count: int,
|
|
2761
|
+
passed_count: int,
|
|
2762
|
+
failed_count: int,
|
|
2763
|
+
score: float,
|
|
2764
|
+
) -> dict[str, Any]:
|
|
2765
|
+
coverage = _suite_trinity_coverage(capabilities)
|
|
2766
|
+
admitted_count = int(evidence_admission.get("admitted_count") or 0)
|
|
2767
|
+
admitted_frozen_count = int(evidence_admission.get("admitted_frozen_count") or 0)
|
|
2768
|
+
governed_count = int(optimizer_governance.get("governed_count") or 0)
|
|
2769
|
+
optimizer_failed_count = int(optimizer_governance.get("failed_count") or 0)
|
|
2770
|
+
optimizer_missing_count = int(optimizer_governance.get("missing_count") or 0)
|
|
2771
|
+
gates = [
|
|
2772
|
+
_trust_gate(
|
|
2773
|
+
"execution",
|
|
2774
|
+
passed=failed_count == 0 and executed_count == job_count and suite_passed,
|
|
2775
|
+
required=True,
|
|
2776
|
+
reason="all declared suite jobs executed and exited successfully",
|
|
2777
|
+
evidence={
|
|
2778
|
+
"job_count": job_count,
|
|
2779
|
+
"executed_count": executed_count,
|
|
2780
|
+
"passed_count": passed_count,
|
|
2781
|
+
"failed_count": failed_count,
|
|
2782
|
+
"score": score,
|
|
2783
|
+
},
|
|
2784
|
+
),
|
|
2785
|
+
_trust_gate(
|
|
2786
|
+
"capability_gate",
|
|
2787
|
+
passed=not missing_capabilities,
|
|
2788
|
+
required=True,
|
|
2789
|
+
reason="declared required capabilities were observed",
|
|
2790
|
+
evidence={"missing_required_capabilities": dict(missing_capabilities)},
|
|
2791
|
+
),
|
|
2792
|
+
_trust_gate(
|
|
2793
|
+
"framework_coverage",
|
|
2794
|
+
passed=int(framework_coverage.get("missing_count") or 0) == 0
|
|
2795
|
+
and int(framework_coverage.get("adapter_conformance_failed_count") or 0)
|
|
2796
|
+
== 0,
|
|
2797
|
+
required=True,
|
|
2798
|
+
reason="required framework coverage and adapter conformance passed",
|
|
2799
|
+
evidence={
|
|
2800
|
+
"observed_count": framework_coverage.get("observed_count"),
|
|
2801
|
+
"required_count": framework_coverage.get("required_count"),
|
|
2802
|
+
"missing_count": framework_coverage.get("missing_count"),
|
|
2803
|
+
"adapter_conformance_failed_count": framework_coverage.get(
|
|
2804
|
+
"adapter_conformance_failed_count"
|
|
2805
|
+
),
|
|
2806
|
+
},
|
|
2807
|
+
),
|
|
2808
|
+
_trust_gate(
|
|
2809
|
+
"evidence_admission",
|
|
2810
|
+
passed=admitted_count > 0
|
|
2811
|
+
and int(evidence_admission.get("rejected_count") or 0) == 0,
|
|
2812
|
+
required=False,
|
|
2813
|
+
reason="at least one child artifact is admitted evidence",
|
|
2814
|
+
evidence={
|
|
2815
|
+
"admitted_count": admitted_count,
|
|
2816
|
+
"rejected_count": evidence_admission.get("rejected_count"),
|
|
2817
|
+
"by_status": evidence_admission.get("by_status"),
|
|
2818
|
+
},
|
|
2819
|
+
),
|
|
2820
|
+
_trust_gate(
|
|
2821
|
+
"evidence_freeze",
|
|
2822
|
+
passed=admitted_count > 0 and admitted_frozen_count == admitted_count,
|
|
2823
|
+
required=False,
|
|
2824
|
+
reason="admitted evidence rows are content-addressed",
|
|
2825
|
+
evidence={
|
|
2826
|
+
"admitted_count": admitted_count,
|
|
2827
|
+
"admitted_frozen_count": admitted_frozen_count,
|
|
2828
|
+
},
|
|
2829
|
+
),
|
|
2830
|
+
_trust_gate(
|
|
2831
|
+
"optimizer_governance",
|
|
2832
|
+
passed=governed_count > 0
|
|
2833
|
+
and optimizer_failed_count == 0
|
|
2834
|
+
and optimizer_missing_count == 0,
|
|
2835
|
+
required=False,
|
|
2836
|
+
reason="optimizer children expose passed governance verdicts",
|
|
2837
|
+
evidence={
|
|
2838
|
+
"target_count": optimizer_governance.get("target_count"),
|
|
2839
|
+
"governed_count": governed_count,
|
|
2840
|
+
"failed_count": optimizer_failed_count,
|
|
2841
|
+
"missing_count": optimizer_missing_count,
|
|
2842
|
+
"warning_count": optimizer_governance.get("warning_count"),
|
|
2843
|
+
},
|
|
2844
|
+
),
|
|
2845
|
+
_trust_gate(
|
|
2846
|
+
"trinity_coverage",
|
|
2847
|
+
passed=all(coverage.values()),
|
|
2848
|
+
required=False,
|
|
2849
|
+
reason="suite covers simulation, evaluation, red-team, and optimization",
|
|
2850
|
+
evidence=coverage,
|
|
2851
|
+
),
|
|
2852
|
+
]
|
|
2853
|
+
failed_gate_ids = [
|
|
2854
|
+
gate["id"] for gate in gates if gate["required"] and not gate["passed"]
|
|
2855
|
+
]
|
|
2856
|
+
conditional_gate_ids = [
|
|
2857
|
+
gate["id"] for gate in gates if not gate["required"] and not gate["passed"]
|
|
2858
|
+
]
|
|
2859
|
+
if not suite_passed or failed_gate_ids:
|
|
2860
|
+
verdict = "rejected"
|
|
2861
|
+
elif conditional_gate_ids:
|
|
2862
|
+
verdict = "conditional"
|
|
2863
|
+
else:
|
|
2864
|
+
verdict = "approved"
|
|
2865
|
+
return {
|
|
2866
|
+
"kind": "agent-learning.suite.trust-certificate.v1",
|
|
2867
|
+
"verdict": verdict,
|
|
2868
|
+
"promotion_ready": verdict == "approved",
|
|
2869
|
+
"assurance_level": _suite_assurance_level(verdict, coverage, governed_count),
|
|
2870
|
+
"subject": {
|
|
2871
|
+
"suite_name": str(suite.get("name") or suite_path.stem),
|
|
2872
|
+
"suite_path": str(suite_path),
|
|
2873
|
+
"suite_version": suite.get("version") or AGENT_LEARNING_SUITE_KIND,
|
|
2874
|
+
"job_count": job_count,
|
|
2875
|
+
},
|
|
2876
|
+
"coverage": coverage,
|
|
2877
|
+
"evidence": {
|
|
2878
|
+
"admitted_count": admitted_count,
|
|
2879
|
+
"admitted_frozen_count": admitted_frozen_count,
|
|
2880
|
+
"optimizer_governed_count": governed_count,
|
|
2881
|
+
"optimizer_failed_count": optimizer_failed_count,
|
|
2882
|
+
"optimizer_missing_count": optimizer_missing_count,
|
|
2883
|
+
"framework_observed_count": framework_coverage.get("observed_count"),
|
|
2884
|
+
"framework_missing_count": framework_coverage.get("missing_count"),
|
|
2885
|
+
},
|
|
2886
|
+
"failed_gate_ids": failed_gate_ids,
|
|
2887
|
+
"conditional_gate_ids": conditional_gate_ids,
|
|
2888
|
+
"reason": _suite_trust_reason(verdict, failed_gate_ids, conditional_gate_ids),
|
|
2889
|
+
"gates": gates,
|
|
2890
|
+
"child_ids": [str(child.get("id") or "") for child in children],
|
|
2891
|
+
}
|
|
2892
|
+
|
|
2893
|
+
|
|
2894
|
+
def _suite_trinity_coverage(capabilities: Mapping[str, Sequence[str]]) -> dict[str, bool]:
|
|
2895
|
+
commands = {_suite_key(command) for command in _as_list(capabilities.get("commands"))}
|
|
2896
|
+
result_kinds = {
|
|
2897
|
+
str(item)
|
|
2898
|
+
for item in _as_list(capabilities.get("result_kinds"))
|
|
2899
|
+
if str(item)
|
|
2900
|
+
}
|
|
2901
|
+
return {
|
|
2902
|
+
"simulation": "run" in commands or "agent-learning.run.v1" in result_kinds,
|
|
2903
|
+
"evaluation": bool(
|
|
2904
|
+
commands & {"eval", "eval_artifact", "eval_task", "optimize_eval"}
|
|
2905
|
+
)
|
|
2906
|
+
or "agent-learning.eval.v1" in result_kinds,
|
|
2907
|
+
"redteam": "redteam" in commands or "agent-learning.redteam.v1" in result_kinds,
|
|
2908
|
+
"optimization": bool(commands & {"optimize", "optimize_eval", "optimize_suite"})
|
|
2909
|
+
or "agent-learning.optimization.v1" in result_kinds
|
|
2910
|
+
or "agent-learning.suite-optimization.v1" in result_kinds,
|
|
2911
|
+
}
|
|
2912
|
+
|
|
2913
|
+
|
|
2914
|
+
def _suite_assurance_level(
|
|
2915
|
+
verdict: str,
|
|
2916
|
+
coverage: Mapping[str, bool],
|
|
2917
|
+
governed_count: int,
|
|
2918
|
+
) -> str:
|
|
2919
|
+
if verdict == "rejected":
|
|
2920
|
+
return "rejected"
|
|
2921
|
+
if all(coverage.values()) and governed_count > 0:
|
|
2922
|
+
return "l3_trinity_governed"
|
|
2923
|
+
if coverage.get("simulation") and coverage.get("evaluation"):
|
|
2924
|
+
return "l2_evaluated_simulation"
|
|
2925
|
+
return "l1_partial_evidence"
|
|
2926
|
+
|
|
2927
|
+
|
|
2928
|
+
def _suite_trust_reason(
|
|
2929
|
+
verdict: str,
|
|
2930
|
+
failed_gate_ids: Sequence[str],
|
|
2931
|
+
conditional_gate_ids: Sequence[str],
|
|
2932
|
+
) -> str:
|
|
2933
|
+
if verdict == "approved":
|
|
2934
|
+
return (
|
|
2935
|
+
"Approved: execution, evidence, framework coverage, red-team, "
|
|
2936
|
+
"simulation, evaluation, optimization, and optimizer governance closed."
|
|
2937
|
+
)
|
|
2938
|
+
if verdict == "rejected":
|
|
2939
|
+
return (
|
|
2940
|
+
"Rejected: required suite gates failed"
|
|
2941
|
+
+ (f" ({', '.join(failed_gate_ids)})." if failed_gate_ids else ".")
|
|
2942
|
+
)
|
|
2943
|
+
return (
|
|
2944
|
+
"Conditional: required gates passed but advisory deployment evidence is "
|
|
2945
|
+
f"incomplete ({', '.join(conditional_gate_ids)})."
|
|
2946
|
+
)
|
|
2947
|
+
|
|
2948
|
+
|
|
2949
|
+
def _trust_gate(
|
|
2950
|
+
gate_id: str,
|
|
2951
|
+
*,
|
|
2952
|
+
passed: bool,
|
|
2953
|
+
required: bool,
|
|
2954
|
+
reason: str,
|
|
2955
|
+
evidence: Mapping[str, Any],
|
|
2956
|
+
) -> dict[str, Any]:
|
|
2957
|
+
return {
|
|
2958
|
+
"id": gate_id,
|
|
2959
|
+
"status": "passed" if passed else "failed" if required else "conditional",
|
|
2960
|
+
"passed": passed,
|
|
2961
|
+
"required": required,
|
|
2962
|
+
"reason": reason,
|
|
2963
|
+
"evidence": copy.deepcopy(dict(evidence)),
|
|
2964
|
+
}
|
|
2965
|
+
|
|
2966
|
+
|
|
2967
|
+
def _suite_framework_coverage(
|
|
2968
|
+
children: Sequence[Mapping[str, Any]],
|
|
2969
|
+
*,
|
|
2970
|
+
required_frameworks: Sequence[str],
|
|
2971
|
+
) -> dict[str, Any]:
|
|
2972
|
+
rows: list[dict[str, Any]] = []
|
|
2973
|
+
for child in children:
|
|
2974
|
+
rows.extend(_suite_framework_rows_for_child(_as_mapping(child)))
|
|
2975
|
+
observed = sorted(
|
|
2976
|
+
{
|
|
2977
|
+
_suite_key(row.get("framework"))
|
|
2978
|
+
for row in rows
|
|
2979
|
+
if _suite_key(row.get("framework"))
|
|
2980
|
+
}
|
|
2981
|
+
)
|
|
2982
|
+
required = sorted(
|
|
2983
|
+
{
|
|
2984
|
+
_suite_key(item)
|
|
2985
|
+
for item in _as_list(required_frameworks)
|
|
2986
|
+
if _suite_key(item)
|
|
2987
|
+
}
|
|
2988
|
+
)
|
|
2989
|
+
missing = sorted(set(required) - set(observed))
|
|
2990
|
+
adapter_failures = [
|
|
2991
|
+
row
|
|
2992
|
+
for row in rows
|
|
2993
|
+
if row.get("adapter_conformance_passed") is False
|
|
2994
|
+
]
|
|
2995
|
+
methods: dict[str, set[str]] = {}
|
|
2996
|
+
input_modes: dict[str, set[str]] = {}
|
|
2997
|
+
modalities: dict[str, set[str]] = {}
|
|
2998
|
+
for row in rows:
|
|
2999
|
+
framework = _suite_key(row.get("framework"))
|
|
3000
|
+
if not framework:
|
|
3001
|
+
continue
|
|
3002
|
+
methods.setdefault(framework, set()).update(
|
|
3003
|
+
_suite_key(item)
|
|
3004
|
+
for item in _as_list(row.get("methods"))
|
|
3005
|
+
if _suite_key(item)
|
|
3006
|
+
)
|
|
3007
|
+
input_modes.setdefault(framework, set()).update(
|
|
3008
|
+
_suite_key(item)
|
|
3009
|
+
for item in _as_list(row.get("input_modes"))
|
|
3010
|
+
if _suite_key(item)
|
|
3011
|
+
)
|
|
3012
|
+
modality = _suite_key(row.get("modality"))
|
|
3013
|
+
if modality:
|
|
3014
|
+
modalities.setdefault(framework, set()).add(modality)
|
|
3015
|
+
return {
|
|
3016
|
+
"kind": "agent-learning.suite.framework-coverage.v1",
|
|
3017
|
+
"observed_frameworks": observed,
|
|
3018
|
+
"required_frameworks": required,
|
|
3019
|
+
"missing_required_frameworks": missing,
|
|
3020
|
+
"observed_count": len(observed),
|
|
3021
|
+
"required_count": len(required),
|
|
3022
|
+
"missing_count": len(missing),
|
|
3023
|
+
"adapter_conformance_failed_count": len(adapter_failures),
|
|
3024
|
+
"adapter_conformance_failed_child_ids": [
|
|
3025
|
+
str(row.get("child_id")) for row in adapter_failures
|
|
3026
|
+
],
|
|
3027
|
+
"methods_by_framework": {
|
|
3028
|
+
key: sorted(values) for key, values in sorted(methods.items())
|
|
3029
|
+
},
|
|
3030
|
+
"input_modes_by_framework": {
|
|
3031
|
+
key: sorted(values) for key, values in sorted(input_modes.items())
|
|
3032
|
+
},
|
|
3033
|
+
"modalities_by_framework": {
|
|
3034
|
+
key: sorted(values) for key, values in sorted(modalities.items())
|
|
3035
|
+
},
|
|
3036
|
+
"rows": rows,
|
|
3037
|
+
}
|
|
3038
|
+
|
|
3039
|
+
|
|
3040
|
+
def _suite_framework_findings(
|
|
3041
|
+
coverage: Mapping[str, Any],
|
|
3042
|
+
) -> list[dict[str, Any]]:
|
|
3043
|
+
findings: list[dict[str, Any]] = []
|
|
3044
|
+
missing = [
|
|
3045
|
+
_suite_key(item)
|
|
3046
|
+
for item in _as_list(coverage.get("missing_required_frameworks"))
|
|
3047
|
+
if _suite_key(item)
|
|
3048
|
+
]
|
|
3049
|
+
if missing:
|
|
3050
|
+
findings.append(
|
|
3051
|
+
{
|
|
3052
|
+
"type": "suite_framework_coverage_missing",
|
|
3053
|
+
"level": "error",
|
|
3054
|
+
"reason": (
|
|
3055
|
+
"Suite framework coverage is missing required framework(s): "
|
|
3056
|
+
f"{', '.join(sorted(missing))}."
|
|
3057
|
+
),
|
|
3058
|
+
"missing": sorted(missing),
|
|
3059
|
+
}
|
|
3060
|
+
)
|
|
3061
|
+
failed = [
|
|
3062
|
+
str(item)
|
|
3063
|
+
for item in _as_list(coverage.get("adapter_conformance_failed_child_ids"))
|
|
3064
|
+
if str(item)
|
|
3065
|
+
]
|
|
3066
|
+
if failed:
|
|
3067
|
+
findings.append(
|
|
3068
|
+
{
|
|
3069
|
+
"type": "suite_framework_adapter_conformance_failed",
|
|
3070
|
+
"level": "error",
|
|
3071
|
+
"reason": (
|
|
3072
|
+
"Suite framework coverage found adapter conformance failures "
|
|
3073
|
+
f"in {len(failed)} child row(s)."
|
|
3074
|
+
),
|
|
3075
|
+
"failed_child_ids": failed,
|
|
3076
|
+
}
|
|
3077
|
+
)
|
|
3078
|
+
return findings
|
|
3079
|
+
|
|
3080
|
+
|
|
3081
|
+
def _suite_framework_rows_for_child(child: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
3082
|
+
rows: list[dict[str, Any]] = []
|
|
3083
|
+
result = _as_mapping(child.get("result"))
|
|
3084
|
+
for nested in _as_list(result.get("children") or result.get("jobs")):
|
|
3085
|
+
nested_child = _as_mapping(nested)
|
|
3086
|
+
if nested_child:
|
|
3087
|
+
rows.extend(_suite_framework_rows_for_child(nested_child))
|
|
3088
|
+
for state in _suite_framework_environment_states(result):
|
|
3089
|
+
row = _suite_framework_row_from_state(child, state)
|
|
3090
|
+
if row:
|
|
3091
|
+
rows.append(row)
|
|
3092
|
+
return rows
|
|
3093
|
+
|
|
3094
|
+
|
|
3095
|
+
def _suite_framework_environment_states(
|
|
3096
|
+
result: Mapping[str, Any],
|
|
3097
|
+
) -> list[dict[str, Any]]:
|
|
3098
|
+
states: list[dict[str, Any]] = []
|
|
3099
|
+
for report in (
|
|
3100
|
+
_as_mapping(result.get("report")),
|
|
3101
|
+
_as_mapping(_as_mapping(result.get("evaluation")).get("report")),
|
|
3102
|
+
):
|
|
3103
|
+
for case in _as_list(report.get("results")):
|
|
3104
|
+
metadata = _as_mapping(_as_mapping(case).get("metadata"))
|
|
3105
|
+
state = _as_mapping(metadata.get("environment_state"))
|
|
3106
|
+
if state:
|
|
3107
|
+
states.append(state)
|
|
3108
|
+
return states
|
|
3109
|
+
|
|
3110
|
+
|
|
3111
|
+
def _suite_framework_row_from_state(
|
|
3112
|
+
child: Mapping[str, Any],
|
|
3113
|
+
state: Mapping[str, Any],
|
|
3114
|
+
) -> dict[str, Any] | None:
|
|
3115
|
+
runtime = _as_mapping(state.get("framework_runtime"))
|
|
3116
|
+
trace = _as_mapping(state.get("framework_trace"))
|
|
3117
|
+
capability = _as_mapping(state.get("framework_capability_matrix"))
|
|
3118
|
+
framework = (
|
|
3119
|
+
runtime.get("framework")
|
|
3120
|
+
or trace.get("framework")
|
|
3121
|
+
or capability.get("framework")
|
|
3122
|
+
)
|
|
3123
|
+
framework_key = _suite_key(framework)
|
|
3124
|
+
if not framework_key:
|
|
3125
|
+
return None
|
|
3126
|
+
runtime_summary = _as_mapping(runtime.get("summary"))
|
|
3127
|
+
trace_spans = [
|
|
3128
|
+
_as_mapping(span)
|
|
3129
|
+
for span in _as_list(trace.get("spans"))
|
|
3130
|
+
if _as_mapping(span)
|
|
3131
|
+
]
|
|
3132
|
+
trace_signals = sorted(
|
|
3133
|
+
{
|
|
3134
|
+
_suite_key(signal)
|
|
3135
|
+
for span in trace_spans
|
|
3136
|
+
for signal in _as_list(span.get("signals"))
|
|
3137
|
+
if _suite_key(signal)
|
|
3138
|
+
}
|
|
3139
|
+
)
|
|
3140
|
+
conformance = _as_mapping(trace.get("adapter_conformance"))
|
|
3141
|
+
conformance_passed = (
|
|
3142
|
+
bool(conformance.get("passed")) if conformance else None
|
|
3143
|
+
)
|
|
3144
|
+
return {
|
|
3145
|
+
"kind": "agent-learning.suite.framework-coverage-row.v1",
|
|
3146
|
+
"child_id": child.get("id"),
|
|
3147
|
+
"child_name": child.get("name"),
|
|
3148
|
+
"command": child.get("command"),
|
|
3149
|
+
"result_kind": child.get("kind"),
|
|
3150
|
+
"framework": framework_key,
|
|
3151
|
+
"modality": _suite_key(runtime.get("modality") or trace.get("modality")),
|
|
3152
|
+
"methods": sorted(
|
|
3153
|
+
{
|
|
3154
|
+
_suite_key(item)
|
|
3155
|
+
for item in _as_list(runtime_summary.get("methods"))
|
|
3156
|
+
if _suite_key(item)
|
|
3157
|
+
}
|
|
3158
|
+
),
|
|
3159
|
+
"input_modes": sorted(
|
|
3160
|
+
{
|
|
3161
|
+
_suite_key(item)
|
|
3162
|
+
for item in _as_list(runtime_summary.get("input_modes"))
|
|
3163
|
+
if _suite_key(item)
|
|
3164
|
+
}
|
|
3165
|
+
),
|
|
3166
|
+
"tool_call_count": int(runtime_summary.get("tool_call_count") or 0),
|
|
3167
|
+
"trace_span_count": len(trace_spans),
|
|
3168
|
+
"trace_signals": trace_signals,
|
|
3169
|
+
"adapter_conformance_passed": conformance_passed,
|
|
3170
|
+
}
|
|
3171
|
+
|
|
3172
|
+
|
|
3173
|
+
def _suite_job_command_counts(suite: Mapping[str, Any]) -> dict[str, int]:
|
|
3174
|
+
counts: dict[str, int] = {}
|
|
3175
|
+
for job in _suite_jobs(suite):
|
|
3176
|
+
command = str(job.get("command") or "unknown")
|
|
3177
|
+
counts[command] = counts.get(command, 0) + 1
|
|
3178
|
+
return counts
|
|
3179
|
+
|
|
3180
|
+
|
|
3181
|
+
def _artifact_action_plan_card(result: Mapping[str, Any]) -> dict[str, Any] | None:
|
|
3182
|
+
optimization = _as_mapping(result.get("optimization"))
|
|
3183
|
+
history = [
|
|
3184
|
+
_as_mapping(item)
|
|
3185
|
+
for item in _as_list(optimization.get("history"))
|
|
3186
|
+
if _as_mapping(item)
|
|
3187
|
+
]
|
|
3188
|
+
candidate_records = [
|
|
3189
|
+
record
|
|
3190
|
+
for item in history
|
|
3191
|
+
for record in _artifact_action_candidate_records(item)
|
|
3192
|
+
]
|
|
3193
|
+
if not candidate_records:
|
|
3194
|
+
return None
|
|
3195
|
+
selected_action_id = _artifact_action_selected_id(optimization, candidate_records)
|
|
3196
|
+
for record in candidate_records:
|
|
3197
|
+
record["selected"] = bool(record.get("action_id") == selected_action_id)
|
|
3198
|
+
selected = next(
|
|
3199
|
+
(
|
|
3200
|
+
record
|
|
3201
|
+
for record in candidate_records
|
|
3202
|
+
if record.get("action_id") == selected_action_id
|
|
3203
|
+
),
|
|
3204
|
+
max(candidate_records, key=lambda record: float(record.get("score") or 0.0)),
|
|
3205
|
+
)
|
|
3206
|
+
return {
|
|
3207
|
+
"kind": "artifact_action_plan",
|
|
3208
|
+
"status": "selected" if selected_action_id else "observed",
|
|
3209
|
+
"source": "agent_learning_suite_optimization",
|
|
3210
|
+
"selected_action_id": selected.get("action_id"),
|
|
3211
|
+
"selected_candidate_id": selected.get("candidate_id"),
|
|
3212
|
+
"selected_score": selected.get("score"),
|
|
3213
|
+
"selection_reason": _artifact_action_selection_reason(selected),
|
|
3214
|
+
"candidate_count": len(candidate_records),
|
|
3215
|
+
"candidate_score_lineage": candidate_records,
|
|
3216
|
+
"search_paths": _as_string_list(result.get("summary", {}).get("search_paths")),
|
|
3217
|
+
"source_manifest_path": optimization.get("source_manifest_path"),
|
|
3218
|
+
}
|
|
3219
|
+
|
|
3220
|
+
|
|
3221
|
+
def _artifact_action_candidate_records(
|
|
3222
|
+
history_item: Mapping[str, Any],
|
|
3223
|
+
) -> list[dict[str, Any]]:
|
|
3224
|
+
report = _as_mapping(history_item.get("report"))
|
|
3225
|
+
records: list[dict[str, Any]] = []
|
|
3226
|
+
for child in _as_list(report.get("children") or report.get("jobs")):
|
|
3227
|
+
child_item = _as_mapping(child)
|
|
3228
|
+
if str(child_item.get("command") or "").replace("-", "_") != "action_run":
|
|
3229
|
+
continue
|
|
3230
|
+
action_result = _as_mapping(child_item.get("result"))
|
|
3231
|
+
action_summary = _as_mapping(action_result.get("summary"))
|
|
3232
|
+
action_id = str(
|
|
3233
|
+
action_summary.get("action_id")
|
|
3234
|
+
or _artifact_action_id_from_patch(history_item)
|
|
3235
|
+
or child_item.get("id")
|
|
3236
|
+
or ""
|
|
3237
|
+
)
|
|
3238
|
+
output_count = int(action_summary.get("output_count") or 0)
|
|
3239
|
+
outputs_written_count = int(action_summary.get("outputs_written_count") or 0)
|
|
3240
|
+
completion = _artifact_action_completion_rate(
|
|
3241
|
+
action_summary,
|
|
3242
|
+
output_count=output_count,
|
|
3243
|
+
outputs_written_count=outputs_written_count,
|
|
3244
|
+
)
|
|
3245
|
+
action_kind = str(action_summary.get("action_kind") or "cli")
|
|
3246
|
+
evidence_denominator = 1.0 if action_kind == "download" else 4.0
|
|
3247
|
+
evidence_depth = round(
|
|
3248
|
+
min(outputs_written_count / evidence_denominator, 1.0),
|
|
3249
|
+
4,
|
|
3250
|
+
)
|
|
3251
|
+
records.append(
|
|
3252
|
+
{
|
|
3253
|
+
"candidate_id": history_item.get("candidate_id"),
|
|
3254
|
+
"action_id": action_id,
|
|
3255
|
+
"action_label": action_summary.get("action_label"),
|
|
3256
|
+
"action_kind": action_kind,
|
|
3257
|
+
"artifact_ref": action_summary.get("artifact_ref"),
|
|
3258
|
+
"source_card_path": action_summary.get("source_card_path"),
|
|
3259
|
+
"score": history_item.get("score"),
|
|
3260
|
+
"action_score": round((0.8 * completion) + (0.2 * evidence_depth), 4),
|
|
3261
|
+
"status": action_result.get("status") or child_item.get("status"),
|
|
3262
|
+
"exit_code": action_result.get("exit_code", child_item.get("exit_code")),
|
|
3263
|
+
"output_count": output_count,
|
|
3264
|
+
"outputs_written_count": outputs_written_count,
|
|
3265
|
+
"output_completion_rate": completion,
|
|
3266
|
+
"evidence_depth": evidence_depth,
|
|
3267
|
+
"outputs_written": list(action_result.get("outputs_written") or []),
|
|
3268
|
+
"outputs": [
|
|
3269
|
+
{
|
|
3270
|
+
"flag": _as_mapping(output).get("flag"),
|
|
3271
|
+
"path": _as_mapping(output).get("path"),
|
|
3272
|
+
"exists": _as_mapping(output).get("exists"),
|
|
3273
|
+
}
|
|
3274
|
+
for output in _as_list(action_result.get("outputs"))
|
|
3275
|
+
if _as_mapping(output)
|
|
3276
|
+
],
|
|
3277
|
+
"command_args": list(action_result.get("command_args") or []),
|
|
3278
|
+
"patch": copy.deepcopy(dict(history_item.get("patch") or {})),
|
|
3279
|
+
}
|
|
3280
|
+
)
|
|
3281
|
+
return records
|
|
3282
|
+
|
|
3283
|
+
|
|
3284
|
+
def _artifact_action_completion_rate(
|
|
3285
|
+
summary: Mapping[str, Any],
|
|
3286
|
+
*,
|
|
3287
|
+
output_count: int,
|
|
3288
|
+
outputs_written_count: int,
|
|
3289
|
+
) -> float:
|
|
3290
|
+
if summary.get("output_completion_rate") is not None:
|
|
3291
|
+
return round(float(summary.get("output_completion_rate") or 0.0), 4)
|
|
3292
|
+
if output_count:
|
|
3293
|
+
return round(outputs_written_count / output_count, 4)
|
|
3294
|
+
return 1.0
|
|
3295
|
+
|
|
3296
|
+
|
|
3297
|
+
def _artifact_action_selected_id(
|
|
3298
|
+
optimization: Mapping[str, Any],
|
|
3299
|
+
candidates: Sequence[Mapping[str, Any]],
|
|
3300
|
+
) -> str | None:
|
|
3301
|
+
best_config = _as_mapping(optimization.get("best_config"))
|
|
3302
|
+
for job in _as_list(best_config.get("jobs")):
|
|
3303
|
+
action_id = _as_mapping(job).get("action_id")
|
|
3304
|
+
if action_id:
|
|
3305
|
+
return str(action_id)
|
|
3306
|
+
if not candidates:
|
|
3307
|
+
return None
|
|
3308
|
+
best = max(candidates, key=lambda record: float(record.get("score") or 0.0))
|
|
3309
|
+
return str(best.get("action_id")) if best.get("action_id") else None
|
|
3310
|
+
|
|
3311
|
+
|
|
3312
|
+
def _artifact_action_id_from_patch(history_item: Mapping[str, Any]) -> str | None:
|
|
3313
|
+
patch = _as_mapping(history_item.get("patch") or history_item.get("candidate_patch"))
|
|
3314
|
+
job = _as_mapping(patch.get("jobs.0"))
|
|
3315
|
+
action_id = job.get("action_id")
|
|
3316
|
+
return str(action_id) if action_id else None
|
|
3317
|
+
|
|
3318
|
+
|
|
3319
|
+
def _artifact_action_selection_reason(selected: Mapping[str, Any]) -> str:
|
|
3320
|
+
action_id = selected.get("action_id") or "selected action"
|
|
3321
|
+
status = selected.get("status") or "unknown"
|
|
3322
|
+
output_count = selected.get("output_count")
|
|
3323
|
+
outputs_written = selected.get("outputs_written_count")
|
|
3324
|
+
completion = selected.get("output_completion_rate")
|
|
3325
|
+
score = selected.get("score")
|
|
3326
|
+
return (
|
|
3327
|
+
f"Selected {action_id} because it finished with status {status}, "
|
|
3328
|
+
f"score {score}, output completion {completion}, and "
|
|
3329
|
+
f"{outputs_written}/{output_count} declared outputs written."
|
|
3330
|
+
)
|
|
3331
|
+
|
|
3332
|
+
|
|
3333
|
+
def _suite_capability_summary(children: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
|
|
3334
|
+
caps: dict[str, set[str]] = {
|
|
3335
|
+
"channels": set(),
|
|
3336
|
+
"child_ids": set(),
|
|
3337
|
+
"commands": set(),
|
|
3338
|
+
"environment_state_keys": set(),
|
|
3339
|
+
"environment_types": set(),
|
|
3340
|
+
"evidence_roles": set(),
|
|
3341
|
+
"evidence_statuses": set(),
|
|
3342
|
+
"frameworks": set(),
|
|
3343
|
+
"metrics": set(),
|
|
3344
|
+
"modalities": set(),
|
|
3345
|
+
"providers": set(),
|
|
3346
|
+
"result_kinds": set(),
|
|
3347
|
+
"search_paths": set(),
|
|
3348
|
+
}
|
|
3349
|
+
for child in children:
|
|
3350
|
+
_add_capability(caps, "child_ids", child.get("id"))
|
|
3351
|
+
_add_capability(caps, "commands", child.get("command"))
|
|
3352
|
+
_add_capability(caps, "result_kinds", child.get("kind"))
|
|
3353
|
+
evidence = _as_mapping(child.get("evidence"))
|
|
3354
|
+
_add_capability(caps, "evidence_roles", evidence.get("role"))
|
|
3355
|
+
_add_capability(caps, "evidence_statuses", evidence.get("status"))
|
|
3356
|
+
result = _as_mapping(child.get("result"))
|
|
3357
|
+
_collect_result_capabilities(result, caps)
|
|
3358
|
+
return {key: sorted(values) for key, values in caps.items()}
|
|
3359
|
+
|
|
3360
|
+
|
|
3361
|
+
def _suite_required_capabilities(suite: Mapping[str, Any]) -> dict[str, list[str]]:
|
|
3362
|
+
raw = (
|
|
3363
|
+
suite.get("required_capabilities")
|
|
3364
|
+
or suite.get("capability_requirements")
|
|
3365
|
+
or suite.get("capabilities_required")
|
|
3366
|
+
or {}
|
|
3367
|
+
)
|
|
3368
|
+
if not isinstance(raw, Mapping):
|
|
3369
|
+
return {}
|
|
3370
|
+
requirements: dict[str, list[str]] = {}
|
|
3371
|
+
for key, values in raw.items():
|
|
3372
|
+
normalized_key = _suite_key(key)
|
|
3373
|
+
if not normalized_key:
|
|
3374
|
+
continue
|
|
3375
|
+
normalized_values = sorted(
|
|
3376
|
+
{
|
|
3377
|
+
_suite_key(value)
|
|
3378
|
+
for value in _as_list(values)
|
|
3379
|
+
if _suite_key(value)
|
|
3380
|
+
}
|
|
3381
|
+
)
|
|
3382
|
+
if normalized_values:
|
|
3383
|
+
requirements[normalized_key] = normalized_values
|
|
3384
|
+
return requirements
|
|
3385
|
+
|
|
3386
|
+
|
|
3387
|
+
def _missing_required_capabilities(
|
|
3388
|
+
required: Mapping[str, Sequence[str]],
|
|
3389
|
+
observed: Mapping[str, Sequence[str]],
|
|
3390
|
+
) -> dict[str, list[str]]:
|
|
3391
|
+
missing: dict[str, list[str]] = {}
|
|
3392
|
+
for key, required_values in required.items():
|
|
3393
|
+
observed_values = {_suite_key(value) for value in _as_list(observed.get(key))}
|
|
3394
|
+
missing_values = sorted(
|
|
3395
|
+
{
|
|
3396
|
+
_suite_key(value)
|
|
3397
|
+
for value in _as_list(required_values)
|
|
3398
|
+
if _suite_key(value) and _suite_key(value) not in observed_values
|
|
3399
|
+
}
|
|
3400
|
+
)
|
|
3401
|
+
if missing_values:
|
|
3402
|
+
missing[key] = missing_values
|
|
3403
|
+
return missing
|
|
3404
|
+
|
|
3405
|
+
|
|
3406
|
+
def _suite_capability_findings(
|
|
3407
|
+
missing_capabilities: Mapping[str, Sequence[str]],
|
|
3408
|
+
) -> list[dict[str, Any]]:
|
|
3409
|
+
findings: list[dict[str, Any]] = []
|
|
3410
|
+
for capability, missing_values in sorted(missing_capabilities.items()):
|
|
3411
|
+
values = sorted(_suite_key(value) for value in missing_values if _suite_key(value))
|
|
3412
|
+
if not values:
|
|
3413
|
+
continue
|
|
3414
|
+
findings.append(
|
|
3415
|
+
{
|
|
3416
|
+
"type": "suite_required_capability_missing",
|
|
3417
|
+
"level": "error",
|
|
3418
|
+
"reason": (
|
|
3419
|
+
f"Missing required suite capability `{capability}`: "
|
|
3420
|
+
f"{', '.join(values)}."
|
|
3421
|
+
),
|
|
3422
|
+
"capability": capability,
|
|
3423
|
+
"missing": values,
|
|
3424
|
+
}
|
|
3425
|
+
)
|
|
3426
|
+
return findings
|
|
3427
|
+
|
|
3428
|
+
|
|
3429
|
+
def _collect_result_capabilities(payload: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
|
|
3430
|
+
for child in _as_list(payload.get("children") or payload.get("jobs")):
|
|
3431
|
+
child_item = _as_mapping(child)
|
|
3432
|
+
if not child_item:
|
|
3433
|
+
continue
|
|
3434
|
+
_add_capability(caps, "child_ids", child_item.get("id"))
|
|
3435
|
+
_add_capability(caps, "commands", child_item.get("command"))
|
|
3436
|
+
_add_capability(caps, "result_kinds", child_item.get("kind"))
|
|
3437
|
+
_collect_result_capabilities(_as_mapping(child_item.get("result")), caps)
|
|
3438
|
+
_collect_summary_capabilities(_as_mapping(payload.get("summary")), caps)
|
|
3439
|
+
optimization = _as_mapping(payload.get("optimization"))
|
|
3440
|
+
best_config = _as_mapping(optimization.get("best_config"))
|
|
3441
|
+
simulation = _as_mapping(best_config.get("simulation"))
|
|
3442
|
+
for environment in _as_list(simulation.get("environments")):
|
|
3443
|
+
env = _as_mapping(environment)
|
|
3444
|
+
_add_capability(caps, "environment_types", env.get("type"))
|
|
3445
|
+
for history in _as_list(optimization.get("history")):
|
|
3446
|
+
item = _as_mapping(history)
|
|
3447
|
+
_add_capabilities(caps, "metrics", _as_mapping(item.get("metrics")).keys())
|
|
3448
|
+
_collect_report_capabilities(_as_mapping(item.get("report")), caps)
|
|
3449
|
+
_collect_report_capabilities(_as_mapping(payload.get("report")), caps)
|
|
3450
|
+
_collect_report_capabilities(_as_mapping(_as_mapping(payload.get("evaluation")).get("report")), caps)
|
|
3451
|
+
_collect_payload_capabilities(payload, caps)
|
|
3452
|
+
|
|
3453
|
+
|
|
3454
|
+
def _collect_report_capabilities(report: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
|
|
3455
|
+
for result in _as_list(report.get("results")):
|
|
3456
|
+
case = _as_mapping(result)
|
|
3457
|
+
metadata = _as_mapping(case.get("metadata"))
|
|
3458
|
+
environment_state = _as_mapping(metadata.get("environment_state"))
|
|
3459
|
+
_add_capabilities(caps, "environment_state_keys", environment_state.keys())
|
|
3460
|
+
for state in environment_state.values():
|
|
3461
|
+
_collect_payload_capabilities(state, caps)
|
|
3462
|
+
_collect_payload_capabilities(_as_mapping(case.get("evaluation")), caps)
|
|
3463
|
+
|
|
3464
|
+
|
|
3465
|
+
def _collect_payload_capabilities(
|
|
3466
|
+
value: Any,
|
|
3467
|
+
caps: dict[str, set[str]],
|
|
3468
|
+
*,
|
|
3469
|
+
depth: int = 0,
|
|
3470
|
+
) -> None:
|
|
3471
|
+
if depth > 12:
|
|
3472
|
+
return
|
|
3473
|
+
if isinstance(value, Mapping):
|
|
3474
|
+
item = _as_mapping(value)
|
|
3475
|
+
_collect_summary_capabilities(_as_mapping(item.get("summary")), caps)
|
|
3476
|
+
_add_capability(caps, "frameworks", item.get("framework"))
|
|
3477
|
+
_add_capability(caps, "providers", item.get("provider"))
|
|
3478
|
+
_add_capability(caps, "providers", item.get("provider_id"))
|
|
3479
|
+
_add_capability(caps, "providers", item.get("provider_type"))
|
|
3480
|
+
_add_capability(caps, "channels", item.get("channel"))
|
|
3481
|
+
_add_capability(caps, "channels", item.get("modality"))
|
|
3482
|
+
_add_capability(caps, "modalities", item.get("modality"))
|
|
3483
|
+
_add_capabilities(caps, "metrics", _as_mapping(item.get("metrics")).keys())
|
|
3484
|
+
if _suite_key(item.get("type")) in _KNOWN_ENVIRONMENT_TYPES:
|
|
3485
|
+
_add_capability(caps, "environment_types", item.get("type"))
|
|
3486
|
+
for metric in _as_list(item.get("metrics")):
|
|
3487
|
+
metric_item = _as_mapping(metric)
|
|
3488
|
+
_add_capability(caps, "metrics", metric_item.get("name"))
|
|
3489
|
+
for child in item.values():
|
|
3490
|
+
_collect_payload_capabilities(child, caps, depth=depth + 1)
|
|
3491
|
+
elif isinstance(value, list):
|
|
3492
|
+
for child in value:
|
|
3493
|
+
_collect_payload_capabilities(child, caps, depth=depth + 1)
|
|
3494
|
+
|
|
3495
|
+
|
|
3496
|
+
def _collect_summary_capabilities(summary: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
|
|
3497
|
+
if not summary:
|
|
3498
|
+
return
|
|
3499
|
+
_add_capabilities(caps, "search_paths", summary.get("search_paths"))
|
|
3500
|
+
_add_capabilities(caps, "providers", summary.get("observed_providers"))
|
|
3501
|
+
_add_capabilities(caps, "providers", summary.get("required_providers"))
|
|
3502
|
+
_add_capabilities(caps, "channels", summary.get("observed_channels"))
|
|
3503
|
+
_add_capabilities(caps, "channels", summary.get("required_channels"))
|
|
3504
|
+
_add_capabilities(caps, "frameworks", summary.get("trace_frameworks"))
|
|
3505
|
+
_add_capabilities(caps, "frameworks", summary.get("observed_frameworks"))
|
|
3506
|
+
_add_capabilities(caps, "frameworks", summary.get("required_trace_frameworks"))
|
|
3507
|
+
_add_capabilities(caps, "frameworks", summary.get("frameworks"))
|
|
3508
|
+
_add_capabilities(caps, "environment_state_keys", summary.get("environment_state_keys"))
|
|
3509
|
+
evidence_admission = _as_mapping(summary.get("evidence_admission"))
|
|
3510
|
+
_add_capabilities(caps, "evidence_statuses", evidence_admission.get("by_status"))
|
|
3511
|
+
_add_capabilities(caps, "evidence_roles", evidence_admission.get("by_role"))
|
|
3512
|
+
_add_capabilities(caps, "metrics", summary.get("observed_metrics"))
|
|
3513
|
+
_add_capabilities(caps, "metrics", summary.get("required_metrics"))
|
|
3514
|
+
_add_capabilities(caps, "metrics", summary.get("eval_metrics"))
|
|
3515
|
+
_add_capabilities(caps, "metrics", _as_mapping(summary.get("metric_averages")).keys())
|
|
3516
|
+
provider_channels = _as_mapping(summary.get("provider_channels"))
|
|
3517
|
+
_add_capabilities(caps, "providers", provider_channels.keys())
|
|
3518
|
+
for channels in provider_channels.values():
|
|
3519
|
+
_add_capabilities(caps, "channels", channels)
|
|
3520
|
+
|
|
3521
|
+
|
|
3522
|
+
def _add_capabilities(
|
|
3523
|
+
caps: dict[str, set[str]],
|
|
3524
|
+
key: str,
|
|
3525
|
+
values: Any,
|
|
3526
|
+
) -> None:
|
|
3527
|
+
if isinstance(values, Mapping):
|
|
3528
|
+
values = values.keys()
|
|
3529
|
+
elif values is None:
|
|
3530
|
+
return
|
|
3531
|
+
elif isinstance(values, (str, bytes)):
|
|
3532
|
+
values = [values]
|
|
3533
|
+
else:
|
|
3534
|
+
try:
|
|
3535
|
+
values = list(values)
|
|
3536
|
+
except TypeError:
|
|
3537
|
+
values = [values]
|
|
3538
|
+
for value in values:
|
|
3539
|
+
_add_capability(caps, key, value)
|
|
3540
|
+
|
|
3541
|
+
|
|
3542
|
+
def _add_capability(caps: dict[str, set[str]], key: str, value: Any) -> None:
|
|
3543
|
+
normalized = _suite_key(value)
|
|
3544
|
+
if normalized:
|
|
3545
|
+
caps[key].add(normalized)
|
|
3546
|
+
|
|
3547
|
+
|
|
3548
|
+
def _suite_findings(children: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
|
3549
|
+
findings: list[dict[str, Any]] = []
|
|
3550
|
+
for child in children:
|
|
3551
|
+
exit_code = int(child.get("exit_code", 1))
|
|
3552
|
+
if exit_code != 0:
|
|
3553
|
+
findings.append(
|
|
3554
|
+
{
|
|
3555
|
+
"type": "suite_child_failed",
|
|
3556
|
+
"level": "error",
|
|
3557
|
+
"reason": (
|
|
3558
|
+
f"{child.get('command')} {child.get('id')} exited "
|
|
3559
|
+
f"{exit_code}."
|
|
3560
|
+
),
|
|
3561
|
+
"job": child.get("id"),
|
|
3562
|
+
"command": child.get("command"),
|
|
3563
|
+
"path": child.get("path"),
|
|
3564
|
+
}
|
|
3565
|
+
)
|
|
3566
|
+
for finding in list(child.get("findings") or []):
|
|
3567
|
+
if isinstance(finding, Mapping):
|
|
3568
|
+
copied = copy.deepcopy(dict(finding))
|
|
3569
|
+
copied.setdefault("job", child.get("id"))
|
|
3570
|
+
copied.setdefault("command", child.get("command"))
|
|
3571
|
+
copied.setdefault("path", child.get("path"))
|
|
3572
|
+
findings.append(copied)
|
|
3573
|
+
return findings
|
|
3574
|
+
|
|
3575
|
+
|
|
3576
|
+
def _suite_sarif_findings(result: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
3577
|
+
findings = []
|
|
3578
|
+
for finding in list(result.get("findings") or []):
|
|
3579
|
+
if isinstance(finding, Mapping):
|
|
3580
|
+
findings.append(copy.deepcopy(dict(finding)))
|
|
3581
|
+
return findings
|
|
3582
|
+
|
|
3583
|
+
|
|
3584
|
+
def _load_child_source(job: Mapping[str, Any], *, base_dir: Path) -> dict[str, Any]:
|
|
3585
|
+
path = _job_path(job, base_dir=base_dir)
|
|
3586
|
+
loaded = _load_json_or_yaml(path)
|
|
3587
|
+
if not isinstance(loaded, Mapping):
|
|
3588
|
+
raise SuiteError(f"suite job source must be an object: {path}")
|
|
3589
|
+
return dict(loaded)
|
|
3590
|
+
|
|
3591
|
+
|
|
3592
|
+
def _suite_jobs(suite: Mapping[str, Any]) -> list[Mapping[str, Any]]:
|
|
3593
|
+
return [dict(job) for job in _as_list(suite.get("jobs"))]
|
|
3594
|
+
|
|
3595
|
+
|
|
3596
|
+
def _job_path(job: Mapping[str, Any], *, base_dir: Path) -> Path:
|
|
3597
|
+
raw = (
|
|
3598
|
+
job.get("path")
|
|
3599
|
+
or job.get("manifest")
|
|
3600
|
+
or job.get("suite")
|
|
3601
|
+
or job.get("file")
|
|
3602
|
+
or job.get("current")
|
|
3603
|
+
or job.get("result")
|
|
3604
|
+
)
|
|
3605
|
+
if not raw:
|
|
3606
|
+
replay_paths = _as_list(job.get("manifests") or job.get("paths"))
|
|
3607
|
+
if replay_paths:
|
|
3608
|
+
raw = replay_paths[0]
|
|
3609
|
+
if not raw:
|
|
3610
|
+
raise SuiteError(f"suite job {job.get('id') or ''} requires path")
|
|
3611
|
+
return _resolve_path(str(raw), base_dir)
|
|
3612
|
+
|
|
3613
|
+
|
|
3614
|
+
def _job_compare_baseline_path(job: Mapping[str, Any], *, base_dir: Path) -> Path:
|
|
3615
|
+
raw = job.get("baseline") or job.get("baseline_path") or job.get("baseline-path")
|
|
3616
|
+
if not raw:
|
|
3617
|
+
raise SuiteError(f"suite compare job {job.get('id') or ''} requires baseline")
|
|
3618
|
+
return _resolve_path(str(raw), base_dir)
|
|
3619
|
+
|
|
3620
|
+
|
|
3621
|
+
def _job_replay_manifest_paths(job: Mapping[str, Any], *, base_dir: Path) -> list[Path]:
|
|
3622
|
+
raw_values = _as_list(
|
|
3623
|
+
job.get("manifests")
|
|
3624
|
+
or job.get("paths")
|
|
3625
|
+
or job.get("path")
|
|
3626
|
+
or job.get("manifest")
|
|
3627
|
+
)
|
|
3628
|
+
paths = [_resolve_path(str(value), base_dir) for value in raw_values if str(value)]
|
|
3629
|
+
if not paths:
|
|
3630
|
+
raise SuiteError(f"suite replay job {job.get('id') or ''} requires manifests")
|
|
3631
|
+
return paths
|
|
3632
|
+
|
|
3633
|
+
|
|
3634
|
+
def _job_optional_path(
|
|
3635
|
+
job: Mapping[str, Any],
|
|
3636
|
+
*,
|
|
3637
|
+
base_dir: Path,
|
|
3638
|
+
keys: Sequence[str],
|
|
3639
|
+
) -> Optional[Path]:
|
|
3640
|
+
for key in keys:
|
|
3641
|
+
raw = job.get(key)
|
|
3642
|
+
if raw not in (None, ""):
|
|
3643
|
+
return _resolve_path(str(raw), base_dir)
|
|
3644
|
+
return None
|
|
3645
|
+
|
|
3646
|
+
|
|
3647
|
+
def _job_action_id(job: Mapping[str, Any]) -> str:
|
|
3648
|
+
raw = (
|
|
3649
|
+
job.get("action_id")
|
|
3650
|
+
or job.get("action-id")
|
|
3651
|
+
or job.get("action")
|
|
3652
|
+
or job.get("actionId")
|
|
3653
|
+
)
|
|
3654
|
+
if raw in (None, ""):
|
|
3655
|
+
raise SuiteError(f"suite action-run job {job.get('id') or ''} requires action_id")
|
|
3656
|
+
return str(raw)
|
|
3657
|
+
|
|
3658
|
+
|
|
3659
|
+
def _job_action_inputs(job: Mapping[str, Any]) -> dict[str, Any]:
|
|
3660
|
+
raw = job.get("inputs") or job.get("action_inputs") or job.get("action-inputs")
|
|
3661
|
+
if raw in (None, ""):
|
|
3662
|
+
return {}
|
|
3663
|
+
if isinstance(raw, Mapping):
|
|
3664
|
+
return dict(raw)
|
|
3665
|
+
parsed: dict[str, Any] = {}
|
|
3666
|
+
for value in _as_list(raw):
|
|
3667
|
+
text = str(value)
|
|
3668
|
+
if "=" not in text:
|
|
3669
|
+
raise SuiteError(f"suite action-run input must be name=value: {text!r}")
|
|
3670
|
+
key, item = text.split("=", 1)
|
|
3671
|
+
if not key.strip():
|
|
3672
|
+
raise SuiteError(f"suite action-run input has empty name: {text!r}")
|
|
3673
|
+
parsed[key.strip()] = item
|
|
3674
|
+
return parsed
|
|
3675
|
+
|
|
3676
|
+
|
|
3677
|
+
def _job_action_artifact_output(job: Mapping[str, Any]) -> Optional[str]:
|
|
3678
|
+
raw = (
|
|
3679
|
+
job.get("artifact_output")
|
|
3680
|
+
or job.get("artifact-output")
|
|
3681
|
+
or job.get("artifact_output_path")
|
|
3682
|
+
or job.get("artifact-output-path")
|
|
3683
|
+
)
|
|
3684
|
+
if raw in (None, ""):
|
|
3685
|
+
return None
|
|
3686
|
+
return str(raw)
|
|
3687
|
+
|
|
3688
|
+
|
|
3689
|
+
def _job_action_cwd(job: Mapping[str, Any], *, base_dir: Path) -> Path:
|
|
3690
|
+
raw = (
|
|
3691
|
+
job.get("cwd")
|
|
3692
|
+
or job.get("working_dir")
|
|
3693
|
+
or job.get("working-dir")
|
|
3694
|
+
or job.get("workdir")
|
|
3695
|
+
)
|
|
3696
|
+
if raw in (None, ""):
|
|
3697
|
+
return base_dir
|
|
3698
|
+
return _resolve_path(str(raw), base_dir)
|
|
3699
|
+
|
|
3700
|
+
|
|
3701
|
+
def _job_output_paths(job: Mapping[str, Any], base_dir: Path) -> dict[str, list[Path]]:
|
|
3702
|
+
outputs: dict[str, list[Path]] = {
|
|
3703
|
+
"json": [],
|
|
3704
|
+
"junit": [],
|
|
3705
|
+
"sarif": [],
|
|
3706
|
+
"markdown": [],
|
|
3707
|
+
}
|
|
3708
|
+
suite_outputs = dict(job.get("outputs") or {})
|
|
3709
|
+
raw_json = [*_as_list(job.get("output")), *_as_list(suite_outputs.get("json"))]
|
|
3710
|
+
raw_junit = _as_list(suite_outputs.get("junit"))
|
|
3711
|
+
raw_sarif = _as_list(suite_outputs.get("sarif"))
|
|
3712
|
+
raw_markdown = [
|
|
3713
|
+
*_as_list(suite_outputs.get("markdown")),
|
|
3714
|
+
*_as_list(suite_outputs.get("md")),
|
|
3715
|
+
]
|
|
3716
|
+
for value in raw_json:
|
|
3717
|
+
path = _resolve_path(str(value), base_dir)
|
|
3718
|
+
if path.name.endswith((".junit.xml", ".xml")):
|
|
3719
|
+
outputs["junit"].append(path)
|
|
3720
|
+
elif path.name.endswith((".sarif", ".sarif.json")):
|
|
3721
|
+
outputs["sarif"].append(path)
|
|
3722
|
+
else:
|
|
3723
|
+
outputs["json"].append(path)
|
|
3724
|
+
outputs["junit"].extend(_resolve_path(str(value), base_dir) for value in raw_junit)
|
|
3725
|
+
outputs["sarif"].extend(_resolve_path(str(value), base_dir) for value in raw_sarif)
|
|
3726
|
+
outputs["markdown"].extend(
|
|
3727
|
+
_resolve_path(str(value), base_dir) for value in raw_markdown
|
|
3728
|
+
)
|
|
3729
|
+
return outputs
|
|
3730
|
+
|
|
3731
|
+
|
|
3732
|
+
def _normalize_command(value: Any) -> str:
|
|
3733
|
+
command = str(value or "").strip().lower().replace("-", "_")
|
|
3734
|
+
aliases = {
|
|
3735
|
+
"simulation": "run",
|
|
3736
|
+
"simulate": "run",
|
|
3737
|
+
"evaluation": "eval",
|
|
3738
|
+
"evalartifact": "eval_artifact",
|
|
3739
|
+
"eval_artifacts": "eval_artifact",
|
|
3740
|
+
"eval_report": "eval_artifact",
|
|
3741
|
+
"eval_reports": "eval_artifact",
|
|
3742
|
+
"artifact_eval": "eval_artifact",
|
|
3743
|
+
"artifact_evaluation": "eval_artifact",
|
|
3744
|
+
"evaltask": "eval_task",
|
|
3745
|
+
"eval_tasks": "eval_task",
|
|
3746
|
+
"eval_evidence": "eval_task",
|
|
3747
|
+
"action": "action_run",
|
|
3748
|
+
"actions": "action_run",
|
|
3749
|
+
"actionrun": "action_run",
|
|
3750
|
+
"run_action": "action_run",
|
|
3751
|
+
"task_eval": "eval_task",
|
|
3752
|
+
"task_evaluation": "eval_task",
|
|
3753
|
+
"task_evidence_eval": "eval_task",
|
|
3754
|
+
"red_team": "redteam",
|
|
3755
|
+
"optimization": "optimize",
|
|
3756
|
+
"optimizeeval": "optimize_eval",
|
|
3757
|
+
"optimizesuite": "optimize_suite",
|
|
3758
|
+
"suite_optimization": "optimize_suite",
|
|
3759
|
+
"suite_optimizer": "optimize_suite",
|
|
3760
|
+
"subsuite": "suite",
|
|
3761
|
+
"sub_suite": "suite",
|
|
3762
|
+
"promotion": "promote_to_regression",
|
|
3763
|
+
"regression_promotion": "promote_to_regression",
|
|
3764
|
+
"promote": "promote_to_regression",
|
|
3765
|
+
"minimize": "shrink",
|
|
3766
|
+
"minimize_counterexample": "shrink",
|
|
3767
|
+
}
|
|
3768
|
+
command = aliases.get(command, command)
|
|
3769
|
+
if command not in _CHILD_COMMANDS:
|
|
3770
|
+
allowed = ", ".join(sorted(_CHILD_COMMANDS))
|
|
3771
|
+
raise SuiteError(f"unsupported suite job command: {command}; expected {allowed}")
|
|
3772
|
+
return command
|
|
3773
|
+
|
|
3774
|
+
|
|
3775
|
+
def _normalize_suite_job(job: Mapping[str, Any], index: int) -> dict[str, Any]:
|
|
3776
|
+
item = copy.deepcopy(dict(job))
|
|
3777
|
+
command = _normalize_command(item.get("command") or item.get("type"))
|
|
3778
|
+
path = item.get("path") or item.get("manifest") or item.get("suite")
|
|
3779
|
+
if path in (None, ""):
|
|
3780
|
+
raise ValueError(f"suite job {index} requires a path")
|
|
3781
|
+
item["command"] = command
|
|
3782
|
+
item["path"] = _suite_path_text(path)
|
|
3783
|
+
item["id"] = str(item.get("id") or item.get("name") or f"{command}-{index}")
|
|
3784
|
+
return item
|
|
3785
|
+
|
|
3786
|
+
|
|
3787
|
+
def _suite_path_text(path: str | Path) -> str:
|
|
3788
|
+
return str(path)
|
|
3789
|
+
|
|
3790
|
+
|
|
3791
|
+
def _suite_local_target_text(target: str | Path, *, base_dir: str | Path = ".") -> str:
|
|
3792
|
+
target_text = str(target)
|
|
3793
|
+
module_name, separator, attribute_path = target_text.partition(":")
|
|
3794
|
+
if (
|
|
3795
|
+
separator
|
|
3796
|
+
and attribute_path
|
|
3797
|
+
and (
|
|
3798
|
+
module_name.endswith(".py")
|
|
3799
|
+
or "/" in module_name
|
|
3800
|
+
or "\\" in module_name
|
|
3801
|
+
)
|
|
3802
|
+
):
|
|
3803
|
+
module_path = Path(module_name).expanduser()
|
|
3804
|
+
if not module_path.is_absolute():
|
|
3805
|
+
module_path = Path(base_dir).expanduser() / module_path
|
|
3806
|
+
return f"{module_path.resolve()}:{attribute_path}"
|
|
3807
|
+
return target_text
|
|
3808
|
+
|
|
3809
|
+
|
|
3810
|
+
def _unique_strings(values: Sequence[Any]) -> list[str]:
|
|
3811
|
+
seen: set[str] = set()
|
|
3812
|
+
result: list[str] = []
|
|
3813
|
+
items = values if isinstance(values, (list, tuple, set)) else _as_list(values)
|
|
3814
|
+
for value in items:
|
|
3815
|
+
text = str(value)
|
|
3816
|
+
if text and text not in seen:
|
|
3817
|
+
seen.add(text)
|
|
3818
|
+
result.append(text)
|
|
3819
|
+
return result
|
|
3820
|
+
|
|
3821
|
+
|
|
3822
|
+
def _optimization_lifecycle_paths(
|
|
3823
|
+
*,
|
|
3824
|
+
optimize_manifest_path: str | Path,
|
|
3825
|
+
workspace_dir: str | Path | None,
|
|
3826
|
+
) -> dict[str, Path]:
|
|
3827
|
+
manifest_path = Path(optimize_manifest_path).expanduser().resolve()
|
|
3828
|
+
if workspace_dir is None:
|
|
3829
|
+
workspace = (
|
|
3830
|
+
manifest_path.parent.parent
|
|
3831
|
+
if manifest_path.parent.name == "manifests"
|
|
3832
|
+
else manifest_path.parent
|
|
3833
|
+
)
|
|
3834
|
+
else:
|
|
3835
|
+
workspace = Path(workspace_dir).expanduser().resolve()
|
|
3836
|
+
artifacts = workspace / "artifacts"
|
|
3837
|
+
regressions = workspace / "regressions"
|
|
3838
|
+
return {
|
|
3839
|
+
"optimize_manifest": manifest_path,
|
|
3840
|
+
"optimization": artifacts / "optimization.json",
|
|
3841
|
+
"optimization_junit": artifacts / "optimization.junit.xml",
|
|
3842
|
+
"optimization_sarif": artifacts / "optimization.sarif.json",
|
|
3843
|
+
"optimization_markdown": artifacts / "optimization.md",
|
|
3844
|
+
"optimization_report": artifacts / "optimization-report.json",
|
|
3845
|
+
"optimization_report_markdown": artifacts / "optimization-report.md",
|
|
3846
|
+
"promotion": artifacts / "promotion.json",
|
|
3847
|
+
"promotion_report": artifacts / "promotion-report.json",
|
|
3848
|
+
"promotion_report_markdown": artifacts / "promotion-report.md",
|
|
3849
|
+
"regression_manifest": regressions / "optimized-regression.json",
|
|
3850
|
+
"replay": artifacts / "replay.json",
|
|
3851
|
+
"replay_junit": artifacts / "replay.junit.xml",
|
|
3852
|
+
"replay_sarif": artifacts / "replay.sarif.json",
|
|
3853
|
+
"replay_markdown": artifacts / "replay.md",
|
|
3854
|
+
"replay_report": artifacts / "replay-report.json",
|
|
3855
|
+
"replay_report_markdown": artifacts / "replay-report.md",
|
|
3856
|
+
}
|
|
3857
|
+
|
|
3858
|
+
|
|
3859
|
+
def _required_env_cli_args(required_env: Sequence[str]) -> list[str]:
|
|
3860
|
+
args: list[str] = []
|
|
3861
|
+
for key in _unique_strings(required_env):
|
|
3862
|
+
args.extend(["--required-env", key])
|
|
3863
|
+
return args
|
|
3864
|
+
|
|
3865
|
+
|
|
3866
|
+
def _lifecycle_step(
|
|
3867
|
+
step_id: str,
|
|
3868
|
+
label: str,
|
|
3869
|
+
command_args: Sequence[Any],
|
|
3870
|
+
*,
|
|
3871
|
+
outputs: Optional[Mapping[str, Any]] = None,
|
|
3872
|
+
) -> dict[str, Any]:
|
|
3873
|
+
step = {
|
|
3874
|
+
"id": step_id,
|
|
3875
|
+
"label": label,
|
|
3876
|
+
"kind": "cli",
|
|
3877
|
+
"command": " ".join(shlex.quote(str(arg)) for arg in command_args),
|
|
3878
|
+
"command_args": [str(arg) for arg in command_args],
|
|
3879
|
+
}
|
|
3880
|
+
if outputs:
|
|
3881
|
+
step["outputs"] = {key: str(value) for key, value in outputs.items()}
|
|
3882
|
+
return step
|
|
3883
|
+
|
|
3884
|
+
|
|
3885
|
+
def _write_lifecycle_result_bundle(
|
|
3886
|
+
result: Mapping[str, Any],
|
|
3887
|
+
*,
|
|
3888
|
+
json_path: Path,
|
|
3889
|
+
junit_path: Path,
|
|
3890
|
+
sarif_path: Path,
|
|
3891
|
+
markdown_path: Path,
|
|
3892
|
+
source_path: Path,
|
|
3893
|
+
) -> list[str]:
|
|
3894
|
+
from fi.alk import simulate
|
|
3895
|
+
|
|
3896
|
+
return [
|
|
3897
|
+
_write_json(json_path, result),
|
|
3898
|
+
_write_text(junit_path, simulate.render_junit(result)),
|
|
3899
|
+
_write_text(sarif_path, simulate.render_sarif(result, manifest_path=source_path)),
|
|
3900
|
+
_write_text(
|
|
3901
|
+
markdown_path,
|
|
3902
|
+
simulate.render_markdown(result, source_path=source_path),
|
|
3903
|
+
),
|
|
3904
|
+
]
|
|
3905
|
+
|
|
3906
|
+
|
|
3907
|
+
def _write_lifecycle_report_bundle(
|
|
3908
|
+
report: Mapping[str, Any],
|
|
3909
|
+
*,
|
|
3910
|
+
json_path: Path,
|
|
3911
|
+
markdown_path: Path,
|
|
3912
|
+
source_path: Path,
|
|
3913
|
+
) -> list[str]:
|
|
3914
|
+
from fi.alk import simulate
|
|
3915
|
+
|
|
3916
|
+
return [
|
|
3917
|
+
_write_json(json_path, report),
|
|
3918
|
+
_write_text(
|
|
3919
|
+
markdown_path,
|
|
3920
|
+
simulate.render_markdown(report, source_path=source_path),
|
|
3921
|
+
),
|
|
3922
|
+
]
|
|
3923
|
+
|
|
3924
|
+
|
|
3925
|
+
def _write_json(path: Path, payload: Mapping[str, Any]) -> str:
|
|
3926
|
+
return _write_text(
|
|
3927
|
+
path,
|
|
3928
|
+
json.dumps(payload, indent=2, sort_keys=True, default=str) + "\n",
|
|
3929
|
+
)
|
|
3930
|
+
|
|
3931
|
+
|
|
3932
|
+
def _write_text(path: Path, value: str) -> str:
|
|
3933
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
3934
|
+
path.write_text(value, encoding="utf-8")
|
|
3935
|
+
return str(path)
|
|
3936
|
+
|
|
3937
|
+
|
|
3938
|
+
def _job_name(job: Mapping[str, Any]) -> Optional[str]:
|
|
3939
|
+
value = job.get("name")
|
|
3940
|
+
if value in (None, ""):
|
|
3941
|
+
return None
|
|
3942
|
+
return str(value)
|
|
3943
|
+
|
|
3944
|
+
|
|
3945
|
+
def _job_threshold(
|
|
3946
|
+
job: Mapping[str, Any],
|
|
3947
|
+
suite_options: SuiteRunOptions,
|
|
3948
|
+
) -> Optional[float]:
|
|
3949
|
+
if job.get("threshold") is not None:
|
|
3950
|
+
return float(job["threshold"])
|
|
3951
|
+
return suite_options.threshold
|
|
3952
|
+
|
|
3953
|
+
|
|
3954
|
+
def _job_max_candidates(
|
|
3955
|
+
job: Mapping[str, Any],
|
|
3956
|
+
suite_options: SuiteRunOptions,
|
|
3957
|
+
) -> Optional[int]:
|
|
3958
|
+
if job.get("max_candidates") is not None:
|
|
3959
|
+
return int(job["max_candidates"])
|
|
3960
|
+
if job.get("max-candidates") is not None:
|
|
3961
|
+
return int(job["max-candidates"])
|
|
3962
|
+
return suite_options.max_candidates
|
|
3963
|
+
|
|
3964
|
+
|
|
3965
|
+
def _job_dry_run(job: Mapping[str, Any], suite_options: SuiteRunOptions) -> bool:
|
|
3966
|
+
return bool(suite_options.dry_run or job.get("dry_run") or job.get("dry-run"))
|
|
3967
|
+
|
|
3968
|
+
|
|
3969
|
+
def _job_int(
|
|
3970
|
+
job: Mapping[str, Any],
|
|
3971
|
+
*keys: str,
|
|
3972
|
+
default: int,
|
|
3973
|
+
) -> int:
|
|
3974
|
+
for key in keys:
|
|
3975
|
+
if job.get(key) is not None:
|
|
3976
|
+
return int(job[key])
|
|
3977
|
+
return default
|
|
3978
|
+
|
|
3979
|
+
|
|
3980
|
+
def _job_float(
|
|
3981
|
+
job: Mapping[str, Any],
|
|
3982
|
+
*keys: str,
|
|
3983
|
+
default: float,
|
|
3984
|
+
) -> float:
|
|
3985
|
+
for key in keys:
|
|
3986
|
+
if job.get(key) is not None:
|
|
3987
|
+
return float(job[key])
|
|
3988
|
+
return default
|
|
3989
|
+
|
|
3990
|
+
|
|
3991
|
+
def _job_optional_float(
|
|
3992
|
+
job: Mapping[str, Any],
|
|
3993
|
+
*keys: str,
|
|
3994
|
+
) -> Optional[float]:
|
|
3995
|
+
for key in keys:
|
|
3996
|
+
if job.get(key) is not None:
|
|
3997
|
+
return float(job[key])
|
|
3998
|
+
return None
|
|
3999
|
+
|
|
4000
|
+
|
|
4001
|
+
def _merge_options(
|
|
4002
|
+
options: Optional[SuiteRunOptions],
|
|
4003
|
+
*,
|
|
4004
|
+
name: Optional[str] = None,
|
|
4005
|
+
threshold: Optional[float] = None,
|
|
4006
|
+
max_candidates: Optional[int] = None,
|
|
4007
|
+
dry_run: Optional[bool] = None,
|
|
4008
|
+
fail_fast: Optional[bool] = None,
|
|
4009
|
+
require_optimizer_governance: Optional[bool] = None,
|
|
4010
|
+
) -> SuiteRunOptions:
|
|
4011
|
+
base = options or SuiteRunOptions()
|
|
4012
|
+
return SuiteRunOptions(
|
|
4013
|
+
name=name if name is not None else base.name,
|
|
4014
|
+
threshold=threshold if threshold is not None else base.threshold,
|
|
4015
|
+
max_candidates=(
|
|
4016
|
+
max_candidates if max_candidates is not None else base.max_candidates
|
|
4017
|
+
),
|
|
4018
|
+
dry_run=dry_run if dry_run is not None else base.dry_run,
|
|
4019
|
+
fail_fast=fail_fast if fail_fast is not None else base.fail_fast,
|
|
4020
|
+
require_optimizer_governance=(
|
|
4021
|
+
require_optimizer_governance
|
|
4022
|
+
if require_optimizer_governance is not None
|
|
4023
|
+
else base.require_optimizer_governance
|
|
4024
|
+
),
|
|
4025
|
+
)
|
|
4026
|
+
|
|
4027
|
+
|
|
4028
|
+
def _merge_optimization_options(
|
|
4029
|
+
options: Optional[SuiteOptimizationOptions],
|
|
4030
|
+
*,
|
|
4031
|
+
name: Optional[str] = None,
|
|
4032
|
+
threshold: Optional[float] = None,
|
|
4033
|
+
max_candidates: Optional[int] = None,
|
|
4034
|
+
dry_run: Optional[bool] = None,
|
|
4035
|
+
) -> SuiteOptimizationOptions:
|
|
4036
|
+
base = options or SuiteOptimizationOptions()
|
|
4037
|
+
return SuiteOptimizationOptions(
|
|
4038
|
+
name=name if name is not None else base.name,
|
|
4039
|
+
threshold=threshold if threshold is not None else base.threshold,
|
|
4040
|
+
max_candidates=(
|
|
4041
|
+
max_candidates if max_candidates is not None else base.max_candidates
|
|
4042
|
+
),
|
|
4043
|
+
dry_run=dry_run if dry_run is not None else base.dry_run,
|
|
4044
|
+
)
|
|
4045
|
+
|
|
4046
|
+
|
|
4047
|
+
def _optimization_cli() -> Any:
|
|
4048
|
+
import importlib
|
|
4049
|
+
|
|
4050
|
+
return importlib.import_module("fi.alk.simulate.cli")
|
|
4051
|
+
|
|
4052
|
+
|
|
4053
|
+
def _load_json_or_yaml(path: Path) -> Any:
|
|
4054
|
+
if path.suffix.lower() in {".yaml", ".yml"}:
|
|
4055
|
+
try:
|
|
4056
|
+
import yaml # type: ignore
|
|
4057
|
+
except Exception as exc: # pragma: no cover - optional dependency clarity
|
|
4058
|
+
raise SuiteError("YAML suite manifests require PyYAML.") from exc
|
|
4059
|
+
with path.open("r", encoding="utf-8") as handle:
|
|
4060
|
+
return yaml.safe_load(handle)
|
|
4061
|
+
with path.open("r", encoding="utf-8") as handle:
|
|
4062
|
+
return json.load(handle)
|
|
4063
|
+
|
|
4064
|
+
|
|
4065
|
+
def _suite_base_dir(suite_path: str | Path) -> Path:
|
|
4066
|
+
path = Path(suite_path).expanduser().resolve()
|
|
4067
|
+
if path.suffix:
|
|
4068
|
+
return path.parent
|
|
4069
|
+
return path
|
|
4070
|
+
|
|
4071
|
+
|
|
4072
|
+
def _resolve_path(value: str, base_dir: Path) -> Path:
|
|
4073
|
+
path = Path(value).expanduser()
|
|
4074
|
+
if path.is_absolute():
|
|
4075
|
+
return path
|
|
4076
|
+
return (base_dir / path).resolve()
|
|
4077
|
+
|
|
4078
|
+
|
|
4079
|
+
def _run_async(awaitable: Any) -> Any:
|
|
4080
|
+
return asyncio.run(awaitable)
|
|
4081
|
+
|
|
4082
|
+
|
|
4083
|
+
def _as_list(value: Any) -> list[Any]:
|
|
4084
|
+
if value is None:
|
|
4085
|
+
return []
|
|
4086
|
+
if isinstance(value, list):
|
|
4087
|
+
return value
|
|
4088
|
+
return [value]
|
|
4089
|
+
|
|
4090
|
+
|
|
4091
|
+
def _as_string_list(value: Any) -> list[str]:
|
|
4092
|
+
return [str(item) for item in _as_list(value) if str(item)]
|
|
4093
|
+
|
|
4094
|
+
|
|
4095
|
+
def _as_mapping(value: Any) -> dict[str, Any]:
|
|
4096
|
+
return dict(value) if isinstance(value, Mapping) else {}
|
|
4097
|
+
|
|
4098
|
+
|
|
4099
|
+
def _suite_key(value: Any) -> str:
|
|
4100
|
+
if isinstance(value, Mapping):
|
|
4101
|
+
return ""
|
|
4102
|
+
if isinstance(value, (list, tuple, set)):
|
|
4103
|
+
return ""
|
|
4104
|
+
return str(value or "").strip().lower().replace("-", "_").replace(" ", "_")
|
|
4105
|
+
|
|
4106
|
+
|
|
4107
|
+
_TRUST_VERDICT_RANK = {
|
|
4108
|
+
"rejected": 0,
|
|
4109
|
+
"conditional": 1,
|
|
4110
|
+
"approved": 2,
|
|
4111
|
+
}
|
|
4112
|
+
|
|
4113
|
+
|
|
4114
|
+
def _optional_bool(value: Any, fallback: Any = None) -> bool | None:
|
|
4115
|
+
candidate = value if value is not None else fallback
|
|
4116
|
+
if candidate is None:
|
|
4117
|
+
return None
|
|
4118
|
+
if isinstance(candidate, bool):
|
|
4119
|
+
return candidate
|
|
4120
|
+
if isinstance(candidate, str):
|
|
4121
|
+
normalized = candidate.strip().lower()
|
|
4122
|
+
if normalized in {"true", "1", "yes"}:
|
|
4123
|
+
return True
|
|
4124
|
+
if normalized in {"false", "0", "no"}:
|
|
4125
|
+
return False
|
|
4126
|
+
return None
|
|
4127
|
+
|
|
4128
|
+
|
|
4129
|
+
_KNOWN_ENVIRONMENT_TYPES = {
|
|
4130
|
+
"adversarial_attack_pack",
|
|
4131
|
+
"agent_control_plane",
|
|
4132
|
+
"agent_integration",
|
|
4133
|
+
"agent_memory_lineage",
|
|
4134
|
+
"agent_trust_boundary",
|
|
4135
|
+
"autonomy_loop",
|
|
4136
|
+
"browser",
|
|
4137
|
+
"domain_package",
|
|
4138
|
+
"framework_capability",
|
|
4139
|
+
"framework_lifecycle",
|
|
4140
|
+
"framework_portability",
|
|
4141
|
+
"framework_probe",
|
|
4142
|
+
"framework_trace",
|
|
4143
|
+
"multimodal_image",
|
|
4144
|
+
"multi_agent_room",
|
|
4145
|
+
"observability_replay",
|
|
4146
|
+
"openenv",
|
|
4147
|
+
"optimizer_trace",
|
|
4148
|
+
"persistent_state_attack",
|
|
4149
|
+
"red_team_campaign",
|
|
4150
|
+
"red_team_readiness",
|
|
4151
|
+
"retrieval_memory",
|
|
4152
|
+
"stateful_tool_world",
|
|
4153
|
+
"streaming_trace",
|
|
4154
|
+
"voice",
|
|
4155
|
+
"workspace_run_manifest",
|
|
4156
|
+
"world_attack_replay",
|
|
4157
|
+
"world_contract",
|
|
4158
|
+
"world_orchestration_replay",
|
|
4159
|
+
}
|
|
4160
|
+
|
|
4161
|
+
|
|
4162
|
+
def _md_cell(value: Any) -> str:
|
|
4163
|
+
return str(value).replace("|", "\\|").replace("\n", " ")
|
|
4164
|
+
|
|
4165
|
+
|
|
4166
|
+
__all__ = [
|
|
4167
|
+
"AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND",
|
|
4168
|
+
"AGENT_LEARNING_SUITE_KIND",
|
|
4169
|
+
"AGENT_LEARNING_SUITE_OPTIMIZATION_KIND",
|
|
4170
|
+
"AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND",
|
|
4171
|
+
"AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND",
|
|
4172
|
+
"SuiteError",
|
|
4173
|
+
"SuiteOptimizationOptions",
|
|
4174
|
+
"SuiteRunOptions",
|
|
4175
|
+
"build_framework_adapter_trinity_suite_optimization_manifest",
|
|
4176
|
+
"build_framework_adapter_trinity_suite_manifest",
|
|
4177
|
+
"build_optimization_lifecycle_plan",
|
|
4178
|
+
"build_regression_artifact_suite_manifest",
|
|
4179
|
+
"build_suite_manifest",
|
|
4180
|
+
"build_trinity_suite_manifest",
|
|
4181
|
+
"load_suite",
|
|
4182
|
+
"load_suite_artifact_file",
|
|
4183
|
+
"load_suite_file",
|
|
4184
|
+
"missing_suite_env",
|
|
4185
|
+
"optimize_suite",
|
|
4186
|
+
"optimize_suite_file",
|
|
4187
|
+
"render_junit",
|
|
4188
|
+
"render_markdown",
|
|
4189
|
+
"render_sarif",
|
|
4190
|
+
"required_suite_env",
|
|
4191
|
+
"run_optimization_lifecycle_file",
|
|
4192
|
+
"run_suite",
|
|
4193
|
+
"run_suite_file",
|
|
4194
|
+
"verify_trust_certificate",
|
|
4195
|
+
"verify_trust_certificate_file",
|
|
4196
|
+
"validate_suite_env",
|
|
4197
|
+
"write_framework_adapter_trinity_suite_optimization_workspace",
|
|
4198
|
+
"write_framework_adapter_trinity_suite_workspace",
|
|
4199
|
+
"write_suite_file",
|
|
4200
|
+
]
|