agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Structured Output Score - Composite Metric.
|
|
3
|
+
|
|
4
|
+
Combines multiple structured validation aspects into a single score.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from typing import Any, Dict, Optional
|
|
8
|
+
|
|
9
|
+
from ..base_metric import BaseMetric
|
|
10
|
+
from .types import StructuredInput, JSONInput, ValidationMode
|
|
11
|
+
from .validators import JSONValidator, YAMLValidator
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class StructuredOutputScore(BaseMetric[StructuredInput]):
|
|
15
|
+
"""
|
|
16
|
+
Comprehensive structured output evaluation.
|
|
17
|
+
|
|
18
|
+
Combines multiple aspects:
|
|
19
|
+
- Syntax validity (parseability)
|
|
20
|
+
- Schema compliance (matches expected structure)
|
|
21
|
+
- Field completeness (required fields present)
|
|
22
|
+
- Type correctness (values have correct types)
|
|
23
|
+
- Value accuracy (optional, if expected provided)
|
|
24
|
+
|
|
25
|
+
Score: 0.0 to 1.0 weighted combination of all aspects.
|
|
26
|
+
|
|
27
|
+
Example:
|
|
28
|
+
>>> metric = StructuredOutputScore()
|
|
29
|
+
>>> result = metric.evaluate([{
|
|
30
|
+
... "response": '{"name": "Alice", "age": 25}',
|
|
31
|
+
... "format": "json",
|
|
32
|
+
... "schema": {
|
|
33
|
+
... "type": "object",
|
|
34
|
+
... "required": ["name", "age"],
|
|
35
|
+
... "properties": {
|
|
36
|
+
... "name": {"type": "string"},
|
|
37
|
+
... "age": {"type": "integer"}
|
|
38
|
+
... }
|
|
39
|
+
... }
|
|
40
|
+
... }])
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def metric_name(self) -> str:
|
|
45
|
+
return "structured_output_score"
|
|
46
|
+
|
|
47
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
48
|
+
super().__init__(config)
|
|
49
|
+
self.json_validator = JSONValidator()
|
|
50
|
+
self._yaml_validator = None
|
|
51
|
+
|
|
52
|
+
# Weights for different aspects
|
|
53
|
+
self.syntax_weight = self.config.get("syntax_weight", 0.2)
|
|
54
|
+
self.schema_weight = self.config.get("schema_weight", 0.3)
|
|
55
|
+
self.completeness_weight = self.config.get("completeness_weight", 0.25)
|
|
56
|
+
self.type_weight = self.config.get("type_weight", 0.15)
|
|
57
|
+
self.value_weight = self.config.get("value_weight", 0.1)
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def yaml_validator(self):
|
|
61
|
+
if self._yaml_validator is None:
|
|
62
|
+
try:
|
|
63
|
+
self._yaml_validator = YAMLValidator()
|
|
64
|
+
except ImportError:
|
|
65
|
+
return None
|
|
66
|
+
return self._yaml_validator
|
|
67
|
+
|
|
68
|
+
def _get_validator(self, format_name: str):
|
|
69
|
+
"""Get validator for format."""
|
|
70
|
+
if format_name.lower() == "yaml":
|
|
71
|
+
if self.yaml_validator is None:
|
|
72
|
+
raise ImportError("PyYAML is required for YAML validation")
|
|
73
|
+
return self.yaml_validator
|
|
74
|
+
return self.json_validator
|
|
75
|
+
|
|
76
|
+
def compute_one(self, inputs: StructuredInput) -> Dict[str, Any]:
|
|
77
|
+
response = inputs.response
|
|
78
|
+
format_name = inputs.format
|
|
79
|
+
schema = inputs.schema
|
|
80
|
+
expected = inputs.expected
|
|
81
|
+
mode_str = inputs.mode
|
|
82
|
+
|
|
83
|
+
# Convert mode string to enum
|
|
84
|
+
try:
|
|
85
|
+
mode = ValidationMode(mode_str)
|
|
86
|
+
except ValueError:
|
|
87
|
+
mode = ValidationMode.COERCE
|
|
88
|
+
|
|
89
|
+
if not response or not response.strip():
|
|
90
|
+
return {
|
|
91
|
+
"output": 0.0,
|
|
92
|
+
"reason": "Empty response",
|
|
93
|
+
"breakdown": {
|
|
94
|
+
"syntax": 0.0,
|
|
95
|
+
"schema": 0.0,
|
|
96
|
+
"completeness": 0.0,
|
|
97
|
+
"types": 0.0,
|
|
98
|
+
"values": 0.0,
|
|
99
|
+
},
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
# Get validator
|
|
103
|
+
try:
|
|
104
|
+
validator = self._get_validator(format_name)
|
|
105
|
+
except ImportError as e:
|
|
106
|
+
return {
|
|
107
|
+
"output": 0.0,
|
|
108
|
+
"reason": str(e),
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
# Initialize scores
|
|
112
|
+
scores = {
|
|
113
|
+
"syntax": 0.0,
|
|
114
|
+
"schema": 0.0,
|
|
115
|
+
"completeness": 0.0,
|
|
116
|
+
"types": 0.0,
|
|
117
|
+
"values": 0.0,
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
# 1. Syntax validation
|
|
121
|
+
syntax_result = validator.validate_syntax(response)
|
|
122
|
+
if not syntax_result.syntax_valid:
|
|
123
|
+
return {
|
|
124
|
+
"output": 0.0,
|
|
125
|
+
"reason": f"Syntax error: {syntax_result.errors[0].message if syntax_result.errors else 'Unknown'}",
|
|
126
|
+
"breakdown": scores,
|
|
127
|
+
"errors": [e.dict() for e in syntax_result.errors],
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
scores["syntax"] = 1.0
|
|
131
|
+
parsed = syntax_result.parsed
|
|
132
|
+
|
|
133
|
+
# Track which dimensions are evaluable
|
|
134
|
+
evaluable = {"syntax": self.syntax_weight}
|
|
135
|
+
|
|
136
|
+
# 2. Schema validation (if schema provided)
|
|
137
|
+
if schema:
|
|
138
|
+
evaluable["schema"] = self.schema_weight
|
|
139
|
+
evaluable["completeness"] = self.completeness_weight
|
|
140
|
+
evaluable["types"] = self.type_weight
|
|
141
|
+
|
|
142
|
+
schema_result = validator.validate_schema(response, schema, mode)
|
|
143
|
+
scores["completeness"] = schema_result.completeness
|
|
144
|
+
|
|
145
|
+
if schema_result.schema_valid:
|
|
146
|
+
scores["schema"] = 1.0
|
|
147
|
+
scores["types"] = 1.0
|
|
148
|
+
else:
|
|
149
|
+
# Analyze errors for partial scores
|
|
150
|
+
type_errors = sum(1 for e in schema_result.errors if e.error_type == "type")
|
|
151
|
+
schema_errors = len(schema_result.errors) - type_errors
|
|
152
|
+
|
|
153
|
+
# Estimate total expected validations
|
|
154
|
+
total_fields = self._count_schema_fields(schema)
|
|
155
|
+
|
|
156
|
+
if total_fields > 0:
|
|
157
|
+
scores["schema"] = max(0.0, 1.0 - schema_errors / total_fields)
|
|
158
|
+
scores["types"] = max(0.0, 1.0 - type_errors / total_fields)
|
|
159
|
+
|
|
160
|
+
# 3. Value accuracy (if expected provided)
|
|
161
|
+
if expected is not None:
|
|
162
|
+
evaluable["values"] = self.value_weight
|
|
163
|
+
|
|
164
|
+
compare_result = validator.compare(response, expected, mode)
|
|
165
|
+
if compare_result.valid:
|
|
166
|
+
scores["values"] = 1.0
|
|
167
|
+
else:
|
|
168
|
+
# Calculate partial value match
|
|
169
|
+
value_errors = len(compare_result.errors)
|
|
170
|
+
total_values = self._count_values(expected)
|
|
171
|
+
scores["values"] = max(0.0, 1.0 - value_errors / max(total_values, 1))
|
|
172
|
+
|
|
173
|
+
# Calculate weighted overall score — only evaluable dimensions contribute
|
|
174
|
+
# Unevaluable dimensions score 0, not 1 (no data ≠ perfect)
|
|
175
|
+
overall = sum(evaluable[k] * scores[k] for k in evaluable)
|
|
176
|
+
|
|
177
|
+
return {
|
|
178
|
+
"output": round(overall, 4),
|
|
179
|
+
"reason": self._generate_reason(scores, evaluable),
|
|
180
|
+
"breakdown": {k: round(v, 4) for k, v in scores.items()},
|
|
181
|
+
"parsed": parsed,
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
def _count_schema_fields(self, schema: Dict[str, Any], depth: int = 0) -> int:
|
|
185
|
+
"""Count fields in schema."""
|
|
186
|
+
if depth > 10:
|
|
187
|
+
return 1
|
|
188
|
+
|
|
189
|
+
count = 0
|
|
190
|
+
if schema.get("type") == "object":
|
|
191
|
+
properties = schema.get("properties", {})
|
|
192
|
+
count = len(properties)
|
|
193
|
+
for prop_schema in properties.values():
|
|
194
|
+
count += self._count_schema_fields(prop_schema, depth + 1)
|
|
195
|
+
elif schema.get("type") == "array":
|
|
196
|
+
items = schema.get("items", {})
|
|
197
|
+
count = 1 + self._count_schema_fields(items, depth + 1)
|
|
198
|
+
|
|
199
|
+
return max(count, 1)
|
|
200
|
+
|
|
201
|
+
def _count_values(self, data: Any) -> int:
|
|
202
|
+
"""Count total values in data structure."""
|
|
203
|
+
if isinstance(data, dict):
|
|
204
|
+
return sum(self._count_values(v) for v in data.values())
|
|
205
|
+
elif isinstance(data, list):
|
|
206
|
+
return sum(self._count_values(item) for item in data)
|
|
207
|
+
else:
|
|
208
|
+
return 1
|
|
209
|
+
|
|
210
|
+
def _generate_reason(self, scores: Dict[str, float], evaluable: Dict[str, float]) -> str:
|
|
211
|
+
"""Generate human-readable reason."""
|
|
212
|
+
issues = []
|
|
213
|
+
if scores["syntax"] < 1.0:
|
|
214
|
+
issues.append("syntax errors")
|
|
215
|
+
if "schema" in evaluable and scores["schema"] < 1.0:
|
|
216
|
+
issues.append("schema violations")
|
|
217
|
+
if "completeness" in evaluable and scores["completeness"] < 1.0:
|
|
218
|
+
issues.append("missing fields")
|
|
219
|
+
if "types" in evaluable and scores["types"] < 1.0:
|
|
220
|
+
issues.append("type errors")
|
|
221
|
+
if "values" in evaluable and scores["values"] < 1.0:
|
|
222
|
+
issues.append("value mismatches")
|
|
223
|
+
|
|
224
|
+
skipped = [k for k in ("schema", "completeness", "types", "values") if k not in evaluable]
|
|
225
|
+
|
|
226
|
+
if not issues and not skipped:
|
|
227
|
+
return "Fully valid structured output"
|
|
228
|
+
parts = []
|
|
229
|
+
if issues:
|
|
230
|
+
parts.append(f"Issues: {', '.join(issues)}")
|
|
231
|
+
if skipped:
|
|
232
|
+
parts.append(f"Not evaluated: {', '.join(skipped)}")
|
|
233
|
+
return "; ".join(parts) if parts else "Fully valid structured output"
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
class QuickStructuredCheck(BaseMetric[JSONInput]):
|
|
237
|
+
"""
|
|
238
|
+
Fast, lightweight structured output check.
|
|
239
|
+
|
|
240
|
+
Quick validation that just checks:
|
|
241
|
+
1. Is it valid JSON?
|
|
242
|
+
2. Does it have the expected keys?
|
|
243
|
+
3. Are the types roughly correct?
|
|
244
|
+
|
|
245
|
+
Score: 0.0 to 1.0
|
|
246
|
+
"""
|
|
247
|
+
|
|
248
|
+
@property
|
|
249
|
+
def metric_name(self) -> str:
|
|
250
|
+
return "quick_structured_check"
|
|
251
|
+
|
|
252
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
253
|
+
super().__init__(config)
|
|
254
|
+
self.validator = JSONValidator()
|
|
255
|
+
|
|
256
|
+
def compute_one(self, inputs: JSONInput) -> Dict[str, Any]:
|
|
257
|
+
response = inputs.response
|
|
258
|
+
schema = inputs.schema
|
|
259
|
+
expected = inputs.expected
|
|
260
|
+
|
|
261
|
+
if not response or not response.strip():
|
|
262
|
+
return {"output": 0.0, "reason": "Empty response"}
|
|
263
|
+
|
|
264
|
+
# Quick syntax check
|
|
265
|
+
syntax_result = self.validator.validate_syntax(response)
|
|
266
|
+
if not syntax_result.syntax_valid:
|
|
267
|
+
return {"output": 0.0, "reason": "Invalid JSON"}
|
|
268
|
+
|
|
269
|
+
parsed = syntax_result.parsed
|
|
270
|
+
score = 0.5 # Base score for valid JSON
|
|
271
|
+
|
|
272
|
+
# Quick schema check
|
|
273
|
+
if schema:
|
|
274
|
+
required = schema.get("required", [])
|
|
275
|
+
if required and isinstance(parsed, dict):
|
|
276
|
+
present = sum(1 for f in required if f in parsed)
|
|
277
|
+
schema_score = present / len(required)
|
|
278
|
+
score = 0.5 + (0.5 * schema_score)
|
|
279
|
+
else:
|
|
280
|
+
score = 1.0
|
|
281
|
+
elif expected is not None:
|
|
282
|
+
# Quick key comparison
|
|
283
|
+
if isinstance(expected, dict) and isinstance(parsed, dict):
|
|
284
|
+
expected_keys = set(expected.keys())
|
|
285
|
+
actual_keys = set(parsed.keys())
|
|
286
|
+
if expected_keys:
|
|
287
|
+
overlap = len(expected_keys & actual_keys) / len(expected_keys)
|
|
288
|
+
score = 0.5 + (0.5 * overlap)
|
|
289
|
+
else:
|
|
290
|
+
score = 1.0
|
|
291
|
+
else:
|
|
292
|
+
score = 1.0 if type(expected) is type(parsed) else 0.5
|
|
293
|
+
|
|
294
|
+
return {
|
|
295
|
+
"output": round(score, 4),
|
|
296
|
+
"reason": "Valid JSON" + (" with expected structure" if score >= 0.8 else ""),
|
|
297
|
+
"parsed": parsed,
|
|
298
|
+
}
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Input types for structured output validation metrics.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import warnings
|
|
6
|
+
from typing import Optional, Dict, Any, List
|
|
7
|
+
from pydantic import BaseModel, Field, ConfigDict
|
|
8
|
+
from enum import Enum
|
|
9
|
+
|
|
10
|
+
warnings.filterwarnings("ignore", message='Field name "schema" in .* shadows an attribute in parent "BaseModel"')
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ValidationMode(Enum):
|
|
14
|
+
"""Validation strictness modes."""
|
|
15
|
+
STRICT = "strict" # Exact match, no extra fields, correct types
|
|
16
|
+
COERCE = "coerce" # Allow type coercion (str -> int, etc.)
|
|
17
|
+
LENIENT = "lenient" # Allow extra fields, flexible types
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class JSONInput(BaseModel):
|
|
21
|
+
"""Input for JSON validation metrics."""
|
|
22
|
+
model_config = ConfigDict(extra="allow")
|
|
23
|
+
|
|
24
|
+
response: str = Field(..., description="LLM-generated JSON string")
|
|
25
|
+
schema: Optional[Dict[str, Any]] = Field(
|
|
26
|
+
None, description="JSON Schema to validate against"
|
|
27
|
+
)
|
|
28
|
+
expected: Optional[Dict[str, Any]] = Field(
|
|
29
|
+
None, description="Expected JSON object for comparison"
|
|
30
|
+
)
|
|
31
|
+
mode: str = Field(
|
|
32
|
+
"coerce", description="Validation strictness: strict, coerce, lenient"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class PydanticInput(BaseModel):
|
|
37
|
+
"""Input for Pydantic model validation."""
|
|
38
|
+
model_config = ConfigDict(extra="allow")
|
|
39
|
+
|
|
40
|
+
response: str = Field(..., description="LLM-generated JSON string")
|
|
41
|
+
model_class: Optional[str] = Field(
|
|
42
|
+
None, description="Fully qualified Pydantic model class name"
|
|
43
|
+
)
|
|
44
|
+
model_schema: Optional[Dict[str, Any]] = Field(
|
|
45
|
+
None, description="JSON Schema derived from Pydantic model"
|
|
46
|
+
)
|
|
47
|
+
mode: str = Field("coerce")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class YAMLInput(BaseModel):
|
|
51
|
+
"""Input for YAML validation metrics."""
|
|
52
|
+
model_config = ConfigDict(extra="allow")
|
|
53
|
+
|
|
54
|
+
response: str = Field(..., description="LLM-generated YAML string")
|
|
55
|
+
schema: Optional[Dict[str, Any]] = Field(
|
|
56
|
+
None, description="JSON Schema to validate against"
|
|
57
|
+
)
|
|
58
|
+
expected: Optional[Dict[str, Any]] = Field(
|
|
59
|
+
None, description="Expected YAML content as dict"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class StructuredInput(BaseModel):
|
|
64
|
+
"""Generic input for any structured format."""
|
|
65
|
+
model_config = ConfigDict(extra="allow")
|
|
66
|
+
|
|
67
|
+
response: str = Field(..., description="LLM-generated structured output")
|
|
68
|
+
format: str = Field("json", description="Format: json, xml, yaml, toml")
|
|
69
|
+
schema: Optional[Dict[str, Any]] = Field(None, description="Schema to validate")
|
|
70
|
+
expected: Optional[Any] = Field(None, description="Expected output")
|
|
71
|
+
mode: str = Field("coerce")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class ValidationError(BaseModel):
|
|
75
|
+
"""Single validation error."""
|
|
76
|
+
path: str = Field(..., description="JSON path to error (e.g., '$.user.name')")
|
|
77
|
+
message: str = Field(..., description="Error description")
|
|
78
|
+
error_type: str = Field(..., description="Error type: syntax, type, missing, extra")
|
|
79
|
+
expected: Optional[Any] = Field(None, description="Expected value/type")
|
|
80
|
+
actual: Optional[Any] = Field(None, description="Actual value/type")
|
|
81
|
+
|
|
82
|
+
def dict(self, **kwargs):
|
|
83
|
+
"""Convert to dictionary."""
|
|
84
|
+
return {
|
|
85
|
+
"path": self.path,
|
|
86
|
+
"message": self.message,
|
|
87
|
+
"error_type": self.error_type,
|
|
88
|
+
"expected": self.expected,
|
|
89
|
+
"actual": self.actual,
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class ValidationResult(BaseModel):
|
|
94
|
+
"""Complete validation result."""
|
|
95
|
+
model_config = ConfigDict(extra="allow")
|
|
96
|
+
|
|
97
|
+
valid: bool = Field(..., description="Overall validity")
|
|
98
|
+
errors: List[ValidationError] = Field(default_factory=list)
|
|
99
|
+
warnings: List[ValidationError] = Field(default_factory=list)
|
|
100
|
+
|
|
101
|
+
# Detailed scores
|
|
102
|
+
syntax_valid: bool = True
|
|
103
|
+
schema_valid: bool = True
|
|
104
|
+
type_valid: bool = True
|
|
105
|
+
completeness: float = 1.0 # 0-1, fraction of required fields present
|
|
106
|
+
|
|
107
|
+
# Parsed output (if successful)
|
|
108
|
+
parsed: Optional[Any] = None
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Validators for structured output formats.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .base import BaseValidator
|
|
6
|
+
from .json_validator import JSONValidator
|
|
7
|
+
from .pydantic_validator import PydanticValidator
|
|
8
|
+
from .yaml_validator import YAMLValidator
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"BaseValidator",
|
|
12
|
+
"JSONValidator",
|
|
13
|
+
"PydanticValidator",
|
|
14
|
+
"YAMLValidator",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
# Validator registry for easy lookup
|
|
18
|
+
VALIDATORS = {
|
|
19
|
+
"json": JSONValidator,
|
|
20
|
+
"yaml": YAMLValidator,
|
|
21
|
+
"pydantic": PydanticValidator,
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def get_validator(format_name: str) -> BaseValidator:
|
|
26
|
+
"""Get validator instance by format name."""
|
|
27
|
+
validator_class = VALIDATORS.get(format_name.lower())
|
|
28
|
+
if validator_class is None:
|
|
29
|
+
raise ValueError(f"Unknown format: {format_name}. Available: {list(VALIDATORS.keys())}")
|
|
30
|
+
return validator_class()
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Base validator interface for structured output validation.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from typing import Any, Dict, List
|
|
7
|
+
from ..types import ValidationResult, ValidationError, ValidationMode
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class BaseValidator(ABC):
|
|
11
|
+
"""Abstract base class for format validators."""
|
|
12
|
+
|
|
13
|
+
format_name: str = "unknown"
|
|
14
|
+
|
|
15
|
+
@abstractmethod
|
|
16
|
+
def validate_syntax(self, content: str) -> ValidationResult:
|
|
17
|
+
"""
|
|
18
|
+
Validate syntax only (is it parseable?).
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
ValidationResult with syntax_valid set
|
|
22
|
+
"""
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
@abstractmethod
|
|
26
|
+
def validate_schema(
|
|
27
|
+
self,
|
|
28
|
+
content: str,
|
|
29
|
+
schema: Dict[str, Any],
|
|
30
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
31
|
+
) -> ValidationResult:
|
|
32
|
+
"""
|
|
33
|
+
Validate against a schema.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
content: Raw string content
|
|
37
|
+
schema: Schema to validate against
|
|
38
|
+
mode: Validation strictness
|
|
39
|
+
|
|
40
|
+
Returns:
|
|
41
|
+
ValidationResult with full validation details
|
|
42
|
+
"""
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
@abstractmethod
|
|
46
|
+
def parse(self, content: str) -> Any:
|
|
47
|
+
"""
|
|
48
|
+
Parse content into Python object.
|
|
49
|
+
|
|
50
|
+
Raises:
|
|
51
|
+
ValueError: If content cannot be parsed
|
|
52
|
+
"""
|
|
53
|
+
pass
|
|
54
|
+
|
|
55
|
+
def compare(
|
|
56
|
+
self,
|
|
57
|
+
content: str,
|
|
58
|
+
expected: Any,
|
|
59
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
60
|
+
) -> ValidationResult:
|
|
61
|
+
"""
|
|
62
|
+
Compare parsed content against expected value.
|
|
63
|
+
|
|
64
|
+
Default implementation - can be overridden.
|
|
65
|
+
"""
|
|
66
|
+
try:
|
|
67
|
+
parsed = self.parse(content)
|
|
68
|
+
except Exception as e:
|
|
69
|
+
return ValidationResult(
|
|
70
|
+
valid=False,
|
|
71
|
+
syntax_valid=False,
|
|
72
|
+
errors=[ValidationError(
|
|
73
|
+
path="$",
|
|
74
|
+
message=str(e),
|
|
75
|
+
error_type="syntax",
|
|
76
|
+
)]
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
errors = self._compare_values(parsed, expected, "$", mode)
|
|
80
|
+
|
|
81
|
+
return ValidationResult(
|
|
82
|
+
valid=len(errors) == 0,
|
|
83
|
+
syntax_valid=True,
|
|
84
|
+
errors=errors,
|
|
85
|
+
parsed=parsed,
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def _compare_values(
|
|
89
|
+
self,
|
|
90
|
+
actual: Any,
|
|
91
|
+
expected: Any,
|
|
92
|
+
path: str,
|
|
93
|
+
mode: ValidationMode,
|
|
94
|
+
) -> List[ValidationError]:
|
|
95
|
+
"""Recursively compare values."""
|
|
96
|
+
errors = []
|
|
97
|
+
|
|
98
|
+
# Handle None cases
|
|
99
|
+
if expected is None and actual is None:
|
|
100
|
+
return errors
|
|
101
|
+
if expected is None or actual is None:
|
|
102
|
+
if expected != actual:
|
|
103
|
+
errors.append(ValidationError(
|
|
104
|
+
path=path,
|
|
105
|
+
message="Value mismatch (None vs non-None)",
|
|
106
|
+
error_type="value",
|
|
107
|
+
expected=expected,
|
|
108
|
+
actual=actual,
|
|
109
|
+
))
|
|
110
|
+
return errors
|
|
111
|
+
|
|
112
|
+
# Type comparison
|
|
113
|
+
if type(actual) is not type(expected):
|
|
114
|
+
if mode == ValidationMode.STRICT:
|
|
115
|
+
errors.append(ValidationError(
|
|
116
|
+
path=path,
|
|
117
|
+
message="Type mismatch",
|
|
118
|
+
error_type="type",
|
|
119
|
+
expected=type(expected).__name__,
|
|
120
|
+
actual=type(actual).__name__,
|
|
121
|
+
))
|
|
122
|
+
return errors
|
|
123
|
+
elif mode == ValidationMode.COERCE:
|
|
124
|
+
# Try to coerce
|
|
125
|
+
try:
|
|
126
|
+
actual = type(expected)(actual)
|
|
127
|
+
except (ValueError, TypeError):
|
|
128
|
+
errors.append(ValidationError(
|
|
129
|
+
path=path,
|
|
130
|
+
message=f"Cannot coerce {type(actual).__name__} to {type(expected).__name__}",
|
|
131
|
+
error_type="type",
|
|
132
|
+
expected=type(expected).__name__,
|
|
133
|
+
actual=type(actual).__name__,
|
|
134
|
+
))
|
|
135
|
+
return errors
|
|
136
|
+
|
|
137
|
+
# Dict comparison
|
|
138
|
+
if isinstance(expected, dict):
|
|
139
|
+
# Check for missing keys
|
|
140
|
+
for key in expected:
|
|
141
|
+
if key not in actual:
|
|
142
|
+
errors.append(ValidationError(
|
|
143
|
+
path=f"{path}.{key}",
|
|
144
|
+
message="Missing required field",
|
|
145
|
+
error_type="missing",
|
|
146
|
+
expected=key,
|
|
147
|
+
))
|
|
148
|
+
else:
|
|
149
|
+
errors.extend(self._compare_values(
|
|
150
|
+
actual[key], expected[key], f"{path}.{key}", mode
|
|
151
|
+
))
|
|
152
|
+
|
|
153
|
+
# Check for extra keys in strict mode
|
|
154
|
+
if mode == ValidationMode.STRICT:
|
|
155
|
+
for key in actual:
|
|
156
|
+
if key not in expected:
|
|
157
|
+
errors.append(ValidationError(
|
|
158
|
+
path=f"{path}.{key}",
|
|
159
|
+
message="Unexpected field",
|
|
160
|
+
error_type="extra",
|
|
161
|
+
actual=key,
|
|
162
|
+
))
|
|
163
|
+
|
|
164
|
+
# List comparison
|
|
165
|
+
elif isinstance(expected, list):
|
|
166
|
+
if len(actual) != len(expected):
|
|
167
|
+
errors.append(ValidationError(
|
|
168
|
+
path=path,
|
|
169
|
+
message="Array length mismatch",
|
|
170
|
+
error_type="length",
|
|
171
|
+
expected=len(expected),
|
|
172
|
+
actual=len(actual),
|
|
173
|
+
))
|
|
174
|
+
else:
|
|
175
|
+
for i, (a, e) in enumerate(zip(actual, expected)):
|
|
176
|
+
errors.extend(self._compare_values(a, e, f"{path}[{i}]", mode))
|
|
177
|
+
|
|
178
|
+
# Scalar comparison
|
|
179
|
+
else:
|
|
180
|
+
if actual != expected:
|
|
181
|
+
errors.append(ValidationError(
|
|
182
|
+
path=path,
|
|
183
|
+
message="Value mismatch",
|
|
184
|
+
error_type="value",
|
|
185
|
+
expected=expected,
|
|
186
|
+
actual=actual,
|
|
187
|
+
))
|
|
188
|
+
|
|
189
|
+
return errors
|