agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""Failure-layer classification for live lanes (R§1 #1, R§3.3).
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only. Strictly ordered, first match wins, evaluated BEFORE
|
|
4
|
+
scoring: lane_infra → framework_runtime → provider → agent_behavior. Only
|
|
5
|
+
``agent_behavior`` scores the agent; ``lane_infra`` voids the row.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import dataclasses
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any, Mapping, Sequence
|
|
13
|
+
|
|
14
|
+
from ._contract import FAILURE_LAYERS
|
|
15
|
+
from ._runner import LaneProcessResult, READY_EVENT_TYPE
|
|
16
|
+
|
|
17
|
+
VERIFICATION_EVENT_TYPE = "verification"
|
|
18
|
+
PROVIDER_ERROR_EVENT_TYPE = "provider_error"
|
|
19
|
+
|
|
20
|
+
_TRACEBACK_FILE = re.compile(r'File "([^"]+)"')
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclasses.dataclass
|
|
24
|
+
class FailureAttribution:
|
|
25
|
+
layer: str # member of FAILURE_LAYERS
|
|
26
|
+
detail: str
|
|
27
|
+
scored: bool # True ONLY for agent_behavior (PRD §4.1)
|
|
28
|
+
|
|
29
|
+
def __post_init__(self) -> None:
|
|
30
|
+
if self.layer not in FAILURE_LAYERS:
|
|
31
|
+
raise ValueError(f"unknown failure layer: {self.layer!r}")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _first_event(
|
|
35
|
+
events: Sequence[Mapping[str, Any]], event_type: str
|
|
36
|
+
) -> Mapping[str, Any] | None:
|
|
37
|
+
for event in events:
|
|
38
|
+
if event.get("type") == event_type:
|
|
39
|
+
return event
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _verification_passed(events: Sequence[Mapping[str, Any]]) -> bool | None:
|
|
44
|
+
"""Last worker-reported verification verdict; None = no verifier evidence
|
|
45
|
+
at all (itself lane_infra — sampling without verification is the
|
|
46
|
+
documented gap, R§1 #5)."""
|
|
47
|
+
|
|
48
|
+
verdict: bool | None = None
|
|
49
|
+
for event in events:
|
|
50
|
+
if event.get("type") != VERIFICATION_EVENT_TYPE:
|
|
51
|
+
continue
|
|
52
|
+
payload = event.get("payload")
|
|
53
|
+
if isinstance(payload, Mapping) and "passed" in payload:
|
|
54
|
+
verdict = bool(payload["passed"])
|
|
55
|
+
return verdict
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _deepest_traceback_file(stderr_tail: str) -> str | None:
|
|
59
|
+
matches = _TRACEBACK_FILE.findall(stderr_tail or "")
|
|
60
|
+
return matches[-1] if matches else None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _framework_package_paths(
|
|
64
|
+
events: Sequence[Mapping[str, Any]],
|
|
65
|
+
) -> list[str]:
|
|
66
|
+
ready = _first_event(events, READY_EVENT_TYPE)
|
|
67
|
+
if not ready:
|
|
68
|
+
return []
|
|
69
|
+
payload = ready.get("payload")
|
|
70
|
+
if not isinstance(payload, Mapping):
|
|
71
|
+
return []
|
|
72
|
+
paths = payload.get("package_paths")
|
|
73
|
+
if not isinstance(paths, (list, tuple)):
|
|
74
|
+
return []
|
|
75
|
+
return [str(path) for path in paths if path]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def attribute_failure(
|
|
79
|
+
process: LaneProcessResult,
|
|
80
|
+
transcript_events: Sequence[Mapping[str, Any]],
|
|
81
|
+
) -> FailureAttribution | None:
|
|
82
|
+
"""HTIR-style layer attribution BEFORE scoring (R§1 #1, R§3.3).
|
|
83
|
+
|
|
84
|
+
Order of classification (first match wins):
|
|
85
|
+
lane_infra — spawn failure, timeout before the worker's
|
|
86
|
+
'framework_ready' event, transcript unreadable;
|
|
87
|
+
row is VOID and auto-quarantined, never scored.
|
|
88
|
+
framework_runtime — nonzero exit with a traceback whose deepest frame
|
|
89
|
+
is inside the framework package (worker stamps
|
|
90
|
+
package paths into the 'lane' channel at boot);
|
|
91
|
+
reported as robustness evidence.
|
|
92
|
+
provider — worker-reported provider error event (HTTP 4xx/5xx,
|
|
93
|
+
auth, rate-limit markers from the provider client).
|
|
94
|
+
agent_behavior — process exited clean but verification failed;
|
|
95
|
+
the ONLY class that scores the agent.
|
|
96
|
+
Returns None when the run passed verification.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
events = list(transcript_events)
|
|
100
|
+
ready = _first_event(events, READY_EVENT_TYPE)
|
|
101
|
+
provider_error = _first_event(events, PROVIDER_ERROR_EVENT_TYPE)
|
|
102
|
+
unreadable = _first_event(events, "transcript_unreadable_line")
|
|
103
|
+
|
|
104
|
+
# --- 1. lane_infra ------------------------------------------------------
|
|
105
|
+
if process.exit_code is None and not process.timed_out:
|
|
106
|
+
return FailureAttribution(
|
|
107
|
+
layer="lane_infra",
|
|
108
|
+
detail=f"spawn failure: {process.stderr_tail or 'no process'}",
|
|
109
|
+
scored=False,
|
|
110
|
+
)
|
|
111
|
+
if process.timed_out:
|
|
112
|
+
phase = "before framework_ready" if ready is None else "after ready"
|
|
113
|
+
return FailureAttribution(
|
|
114
|
+
layer="lane_infra",
|
|
115
|
+
detail=f"timeout {phase} (budget kill)",
|
|
116
|
+
scored=False,
|
|
117
|
+
)
|
|
118
|
+
if unreadable is not None:
|
|
119
|
+
return FailureAttribution(
|
|
120
|
+
layer="lane_infra",
|
|
121
|
+
detail="transcript unreadable",
|
|
122
|
+
scored=False,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
# --- 2./3. crashed worker: framework_runtime, else provider, else infra --
|
|
126
|
+
if process.exit_code not in (0, None):
|
|
127
|
+
deepest = _deepest_traceback_file(process.stderr_tail)
|
|
128
|
+
package_paths = _framework_package_paths(events)
|
|
129
|
+
if deepest and any(deepest.startswith(path) for path in package_paths):
|
|
130
|
+
return FailureAttribution(
|
|
131
|
+
layer="framework_runtime",
|
|
132
|
+
detail=(
|
|
133
|
+
f"worker exit {process.exit_code}; deepest frame "
|
|
134
|
+
f"{deepest} is inside the framework package"
|
|
135
|
+
),
|
|
136
|
+
scored=False,
|
|
137
|
+
)
|
|
138
|
+
if provider_error is not None:
|
|
139
|
+
payload = provider_error.get("payload")
|
|
140
|
+
return FailureAttribution(
|
|
141
|
+
layer="provider",
|
|
142
|
+
detail=f"provider error: {dict(payload) if isinstance(payload, Mapping) else payload}",
|
|
143
|
+
scored=False,
|
|
144
|
+
)
|
|
145
|
+
return FailureAttribution(
|
|
146
|
+
layer="lane_infra",
|
|
147
|
+
detail=(
|
|
148
|
+
f"worker exit {process.exit_code} outside the framework "
|
|
149
|
+
"package (lane/worker fault)"
|
|
150
|
+
),
|
|
151
|
+
scored=False,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
# --- clean exit ----------------------------------------------------------
|
|
155
|
+
if ready is None:
|
|
156
|
+
return FailureAttribution(
|
|
157
|
+
layer="lane_infra",
|
|
158
|
+
detail="worker exited clean but never emitted framework_ready",
|
|
159
|
+
scored=False,
|
|
160
|
+
)
|
|
161
|
+
verification = _verification_passed(events)
|
|
162
|
+
if verification is True:
|
|
163
|
+
return None
|
|
164
|
+
if provider_error is not None:
|
|
165
|
+
payload = provider_error.get("payload")
|
|
166
|
+
return FailureAttribution(
|
|
167
|
+
layer="provider",
|
|
168
|
+
detail=f"provider error: {dict(payload) if isinstance(payload, Mapping) else payload}",
|
|
169
|
+
scored=False,
|
|
170
|
+
)
|
|
171
|
+
if verification is None:
|
|
172
|
+
return FailureAttribution(
|
|
173
|
+
layer="lane_infra",
|
|
174
|
+
detail=(
|
|
175
|
+
"no verifier evidence: repeat carried no programmatic/judge/"
|
|
176
|
+
"end-state verdict (R§1 #5)"
|
|
177
|
+
),
|
|
178
|
+
scored=False,
|
|
179
|
+
)
|
|
180
|
+
return FailureAttribution(
|
|
181
|
+
layer="agent_behavior",
|
|
182
|
+
detail="verification failed on a clean run",
|
|
183
|
+
scored=True,
|
|
184
|
+
)
|
fi/alk/live/_capture.py
ADDED
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""Live→fixture demotion (unit 2.7) + the ONE provenance schema.
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only (plus kit substrate). A captured fixture earns the
|
|
4
|
+
``captured_fixture`` class only after: credential-free green replay, a clean
|
|
5
|
+
secret scrub, the complete provenance block, and a recorded HUMAN review —
|
|
6
|
+
promotion into ``examples/captured/<lane>/`` is a review step, never
|
|
7
|
+
automatic. Candidates stay under the run's artifacts dir with
|
|
8
|
+
``reviewed: false`` (``captured_fixture_candidate`` is NOT an evidence class).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import datetime as _datetime
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Mapping, Sequence
|
|
19
|
+
|
|
20
|
+
from ._attribution import VERIFICATION_EVENT_TYPE
|
|
21
|
+
from ._contract import AGENT_LEARNING_RUN_KIND
|
|
22
|
+
from ._stats import LaneRunResult
|
|
23
|
+
from ._transcript import read_transcript
|
|
24
|
+
|
|
25
|
+
# The ONE provenance schema (identical in PRD §4.1 / ARCH §2c / UI-UX §3).
|
|
26
|
+
CAPTURE_PROVENANCE_FIELDS = (
|
|
27
|
+
"captured_from_lane",
|
|
28
|
+
"captured_run_id",
|
|
29
|
+
"rung",
|
|
30
|
+
"framework",
|
|
31
|
+
"framework_version",
|
|
32
|
+
"capture_date",
|
|
33
|
+
"transcript_sha256",
|
|
34
|
+
"redaction",
|
|
35
|
+
"reviewed",
|
|
36
|
+
"reviewer",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
FIXTURE_CAPTURE_INCOMPLETE_FINDING = "fixture_capture_incomplete_transcript"
|
|
40
|
+
|
|
41
|
+
_CAPTURE_TREE_MARKER = ("examples", "captured")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class CaptureRefusedError(RuntimeError):
|
|
45
|
+
"""Demotion refused; carries the structured finding the CLI surfaces."""
|
|
46
|
+
|
|
47
|
+
def __init__(self, message: str, *, finding: Mapping[str, Any]) -> None:
|
|
48
|
+
super().__init__(message)
|
|
49
|
+
self.finding = dict(finding)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _refuse(detail: str) -> CaptureRefusedError:
|
|
53
|
+
return CaptureRefusedError(
|
|
54
|
+
detail,
|
|
55
|
+
finding={
|
|
56
|
+
"type": FIXTURE_CAPTURE_INCOMPLETE_FINDING,
|
|
57
|
+
"level": "error",
|
|
58
|
+
"detail": detail,
|
|
59
|
+
},
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _canonical_transcript_sha256(events: Sequence[Mapping[str, Any]]) -> str:
|
|
64
|
+
"""Content hash over the canonical JSONL re-serialization of the events —
|
|
65
|
+
replay-verifiable from the embedded transcript alone."""
|
|
66
|
+
|
|
67
|
+
digest = hashlib.sha256()
|
|
68
|
+
for event in events:
|
|
69
|
+
line = json.dumps(dict(event), ensure_ascii=False, default=str) + "\n"
|
|
70
|
+
digest.update(line.encode("utf-8"))
|
|
71
|
+
return digest.hexdigest()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _scrub_values_found(
|
|
75
|
+
serialized: str, required_env: Sequence[str]
|
|
76
|
+
) -> list[str]:
|
|
77
|
+
"""Second scan at capture time (ARCH §2c rule 2): any declared env name
|
|
78
|
+
whose CURRENT value appears in the serialized fixture is a scrub hit."""
|
|
79
|
+
|
|
80
|
+
hits: list[str] = []
|
|
81
|
+
for name in required_env:
|
|
82
|
+
value = os.environ.get(name)
|
|
83
|
+
if value and value in serialized:
|
|
84
|
+
hits.append(name)
|
|
85
|
+
return hits
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _inside_capture_tree(path: Path) -> bool:
|
|
89
|
+
parts = [part.lower() for part in path.resolve().parts]
|
|
90
|
+
for index in range(len(parts) - 1):
|
|
91
|
+
if (parts[index], parts[index + 1]) == _CAPTURE_TREE_MARKER:
|
|
92
|
+
return True
|
|
93
|
+
return False
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _pick_source_repeat(
|
|
97
|
+
result: LaneRunResult, repeat_index: int | None
|
|
98
|
+
) -> Mapping[str, Any]:
|
|
99
|
+
rows = result.per_repeat
|
|
100
|
+
if repeat_index is not None:
|
|
101
|
+
for row in rows:
|
|
102
|
+
if row.get("repeat") == repeat_index:
|
|
103
|
+
return row
|
|
104
|
+
raise ValueError(f"no repeat {repeat_index} in this lane run")
|
|
105
|
+
for row in rows:
|
|
106
|
+
if row.get("passed") and not row.get("quarantined"):
|
|
107
|
+
return row
|
|
108
|
+
raise _refuse(
|
|
109
|
+
"no passing, non-quarantined repeat to capture — a fixture must "
|
|
110
|
+
"replay green, so a non-green run cannot demote"
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def capture_to_fixture(
|
|
115
|
+
result: LaneRunResult,
|
|
116
|
+
*,
|
|
117
|
+
output: Path | str,
|
|
118
|
+
reviewed_by: str | None = None,
|
|
119
|
+
scenario: Mapping[str, Any] | None = None,
|
|
120
|
+
repeat_index: int | None = None,
|
|
121
|
+
) -> Path:
|
|
122
|
+
"""Demote a live run into a credential-free fixture: transcript verbatim
|
|
123
|
+
(already redacted at record time), required_env reduced to NAMES, plus
|
|
124
|
+
the ONE provenance block.
|
|
125
|
+
|
|
126
|
+
Without reviewed_by → a CANDIDATE: written under the run's artifacts dir
|
|
127
|
+
(never a gate-scanned tree), evidence_class kept from the source run,
|
|
128
|
+
reviewed=False. With reviewed_by → runs the credential-free replay
|
|
129
|
+
itself, REFUSES to stamp on a non-green replay, then writes
|
|
130
|
+
evidence_class="captured_fixture", reviewed=True, reviewer=<name> — the
|
|
131
|
+
only form legal under examples/captured/<lane>/. Refuses
|
|
132
|
+
(fixture_capture_incomplete_transcript) when transcript.complete is
|
|
133
|
+
False or the scrub finds any credential value."""
|
|
134
|
+
|
|
135
|
+
output_path = Path(output)
|
|
136
|
+
if reviewed_by is None and _inside_capture_tree(output_path):
|
|
137
|
+
raise _refuse(
|
|
138
|
+
"candidates never land in the gate-scanned capture tree "
|
|
139
|
+
"(examples/captured/); promotion is a human review step — "
|
|
140
|
+
"pass reviewed_by after review"
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
row = _pick_source_repeat(result, repeat_index)
|
|
144
|
+
if not row.get("transcript_complete", False):
|
|
145
|
+
raise _refuse(
|
|
146
|
+
"transcript is truncated (complete: false) — a truncated "
|
|
147
|
+
"transcript can never demote (PRD §4.1)"
|
|
148
|
+
)
|
|
149
|
+
transcript_path = row.get("transcript_path")
|
|
150
|
+
if not transcript_path or not Path(str(transcript_path)).is_file():
|
|
151
|
+
raise _refuse(f"transcript file missing: {transcript_path!r}")
|
|
152
|
+
events = read_transcript(str(transcript_path))
|
|
153
|
+
if not events:
|
|
154
|
+
raise _refuse("transcript is empty — nothing to demote")
|
|
155
|
+
|
|
156
|
+
transcript_sha256 = _canonical_transcript_sha256(events)
|
|
157
|
+
capture_block: dict[str, Any] = {
|
|
158
|
+
"captured_from_lane": result.lane,
|
|
159
|
+
"captured_run_id": result.run_id,
|
|
160
|
+
"rung": result.rung,
|
|
161
|
+
"framework": result.framework,
|
|
162
|
+
"framework_version": result.framework_version,
|
|
163
|
+
"capture_date": _datetime.date.today().isoformat(),
|
|
164
|
+
"transcript_sha256": transcript_sha256,
|
|
165
|
+
"redaction": {
|
|
166
|
+
"required_env_names": list(result.required_env),
|
|
167
|
+
"values_found": 0,
|
|
168
|
+
},
|
|
169
|
+
"reviewed": False,
|
|
170
|
+
"reviewer": None,
|
|
171
|
+
}
|
|
172
|
+
payload: dict[str, Any] = {
|
|
173
|
+
"kind": AGENT_LEARNING_RUN_KIND,
|
|
174
|
+
"name": f"captured-{result.lane}-{result.run_id[:8]}",
|
|
175
|
+
"evidence_class": result.evidence_class, # candidate keeps source class
|
|
176
|
+
"required_env": list(result.required_env),
|
|
177
|
+
"transcript": [dict(event) for event in events],
|
|
178
|
+
"live_lane": {
|
|
179
|
+
"lane": result.lane,
|
|
180
|
+
"rung": result.rung,
|
|
181
|
+
"verdict": result.verdict,
|
|
182
|
+
"captured_repeat": row.get("repeat"),
|
|
183
|
+
"icc": result.icc,
|
|
184
|
+
"divergence_step": result.divergence_step,
|
|
185
|
+
},
|
|
186
|
+
"capture": capture_block,
|
|
187
|
+
}
|
|
188
|
+
if scenario is not None:
|
|
189
|
+
payload["scenario"] = dict(scenario)
|
|
190
|
+
|
|
191
|
+
serialized = json.dumps(payload, ensure_ascii=False, default=str)
|
|
192
|
+
scrub_hits = _scrub_values_found(serialized, result.required_env)
|
|
193
|
+
if scrub_hits:
|
|
194
|
+
raise _refuse(
|
|
195
|
+
"credential values found in the capture for declared env names "
|
|
196
|
+
f"{scrub_hits} — fix the substrate redaction and regenerate; "
|
|
197
|
+
"never hand-edit the secret out"
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
if reviewed_by is not None:
|
|
201
|
+
replay = replay_fixture_payload(payload)
|
|
202
|
+
if replay["verdict"] != "pass":
|
|
203
|
+
raise _refuse(
|
|
204
|
+
"credential-free replay was not green "
|
|
205
|
+
f"({replay['verdict']}; checks: {replay['checks']}) — "
|
|
206
|
+
"refusing to stamp captured_fixture; nothing written"
|
|
207
|
+
)
|
|
208
|
+
payload["evidence_class"] = "captured_fixture"
|
|
209
|
+
capture_block["reviewed"] = True
|
|
210
|
+
capture_block["reviewer"] = str(reviewed_by)
|
|
211
|
+
|
|
212
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
213
|
+
output_path.write_text(
|
|
214
|
+
json.dumps(payload, ensure_ascii=False, indent=2, default=str) + "\n",
|
|
215
|
+
encoding="utf-8",
|
|
216
|
+
)
|
|
217
|
+
return output_path
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def replay_fixture_payload(payload: Mapping[str, Any]) -> dict[str, Any]:
|
|
221
|
+
"""Credential-free replay of a fixture payload: the fixture is data —
|
|
222
|
+
integrity (canonical sha), completeness, redaction residue, and the
|
|
223
|
+
recorded verifier evidence are re-derived; nothing live executes."""
|
|
224
|
+
|
|
225
|
+
checks: dict[str, bool] = {}
|
|
226
|
+
events = payload.get("transcript")
|
|
227
|
+
events = events if isinstance(events, list) else []
|
|
228
|
+
checks["transcript_present"] = bool(events)
|
|
229
|
+
|
|
230
|
+
capture = payload.get("capture")
|
|
231
|
+
capture = capture if isinstance(capture, Mapping) else {}
|
|
232
|
+
expected_sha = capture.get("transcript_sha256")
|
|
233
|
+
checks["transcript_sha256_match"] = bool(events) and (
|
|
234
|
+
_canonical_transcript_sha256(events) == expected_sha
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
verification: bool | None = None
|
|
238
|
+
for event in events:
|
|
239
|
+
if isinstance(event, Mapping) and event.get("type") == VERIFICATION_EVENT_TYPE:
|
|
240
|
+
event_payload = event.get("payload")
|
|
241
|
+
if isinstance(event_payload, Mapping) and "passed" in event_payload:
|
|
242
|
+
verification = bool(event_payload["passed"])
|
|
243
|
+
checks["verification_passed"] = verification is True
|
|
244
|
+
|
|
245
|
+
required_env = [
|
|
246
|
+
str(name) for name in payload.get("required_env") or [] if name
|
|
247
|
+
]
|
|
248
|
+
serialized = json.dumps(dict(payload), ensure_ascii=False, default=str)
|
|
249
|
+
checks["redaction_clean"] = not _scrub_values_found(serialized, required_env)
|
|
250
|
+
|
|
251
|
+
return {
|
|
252
|
+
"verdict": "pass" if all(checks.values()) else "fail",
|
|
253
|
+
"evidence_class": payload.get("evidence_class"),
|
|
254
|
+
"checks": checks,
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def replay_fixture(path: Path | str) -> dict[str, Any]:
|
|
259
|
+
"""Load a fixture file and replay it credential-free (unit 5.4 contract)."""
|
|
260
|
+
|
|
261
|
+
payload = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
262
|
+
if not isinstance(payload, Mapping):
|
|
263
|
+
raise ValueError(f"fixture {path} is not a JSON object")
|
|
264
|
+
return replay_fixture_payload(payload)
|