agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""
|
|
2
|
+
JSON validation with JSON Schema support.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from typing import Any, Dict
|
|
7
|
+
from ..types import ValidationResult, ValidationError, ValidationMode
|
|
8
|
+
from .base import BaseValidator
|
|
9
|
+
|
|
10
|
+
# Optional jsonschema import
|
|
11
|
+
try:
|
|
12
|
+
import jsonschema # noqa: F401
|
|
13
|
+
from jsonschema import Draft7Validator
|
|
14
|
+
_JSONSCHEMA_AVAILABLE = True
|
|
15
|
+
except ImportError:
|
|
16
|
+
_JSONSCHEMA_AVAILABLE = False
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class JSONValidator(BaseValidator):
|
|
20
|
+
"""
|
|
21
|
+
Validator for JSON output.
|
|
22
|
+
|
|
23
|
+
Features:
|
|
24
|
+
- Syntax validation (json.loads)
|
|
25
|
+
- JSON Schema validation (jsonschema library)
|
|
26
|
+
- Deep comparison with expected values
|
|
27
|
+
- Type coercion support
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
format_name = "json"
|
|
31
|
+
|
|
32
|
+
def validate_syntax(self, content: str) -> ValidationResult:
|
|
33
|
+
"""Check if content is valid JSON."""
|
|
34
|
+
try:
|
|
35
|
+
parsed = json.loads(content)
|
|
36
|
+
return ValidationResult(
|
|
37
|
+
valid=True,
|
|
38
|
+
syntax_valid=True,
|
|
39
|
+
parsed=parsed,
|
|
40
|
+
)
|
|
41
|
+
except json.JSONDecodeError as e:
|
|
42
|
+
return ValidationResult(
|
|
43
|
+
valid=False,
|
|
44
|
+
syntax_valid=False,
|
|
45
|
+
errors=[ValidationError(
|
|
46
|
+
path=f"$.char[{e.pos}]",
|
|
47
|
+
message=f"JSON syntax error: {e.msg}",
|
|
48
|
+
error_type="syntax",
|
|
49
|
+
)]
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
def validate_schema(
|
|
53
|
+
self,
|
|
54
|
+
content: str,
|
|
55
|
+
schema: Dict[str, Any],
|
|
56
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
57
|
+
) -> ValidationResult:
|
|
58
|
+
"""Validate JSON against JSON Schema."""
|
|
59
|
+
# First check syntax
|
|
60
|
+
syntax_result = self.validate_syntax(content)
|
|
61
|
+
if not syntax_result.syntax_valid:
|
|
62
|
+
return syntax_result
|
|
63
|
+
|
|
64
|
+
parsed = syntax_result.parsed
|
|
65
|
+
|
|
66
|
+
if not _JSONSCHEMA_AVAILABLE:
|
|
67
|
+
# Fallback to basic type checking
|
|
68
|
+
return self._validate_schema_basic(parsed, schema, mode)
|
|
69
|
+
|
|
70
|
+
# Use jsonschema for validation
|
|
71
|
+
errors = []
|
|
72
|
+
validator = Draft7Validator(schema)
|
|
73
|
+
|
|
74
|
+
for error in validator.iter_errors(parsed):
|
|
75
|
+
path = "$" + "".join(
|
|
76
|
+
f".{p}" if isinstance(p, str) else f"[{p}]"
|
|
77
|
+
for p in error.absolute_path
|
|
78
|
+
)
|
|
79
|
+
errors.append(ValidationError(
|
|
80
|
+
path=path,
|
|
81
|
+
message=error.message,
|
|
82
|
+
error_type=self._classify_schema_error(error),
|
|
83
|
+
expected=error.schema.get("type") if hasattr(error, 'schema') else None,
|
|
84
|
+
actual=type(error.instance).__name__ if error.instance is not None else None,
|
|
85
|
+
))
|
|
86
|
+
|
|
87
|
+
# Calculate completeness
|
|
88
|
+
required_fields = schema.get("required", [])
|
|
89
|
+
if required_fields and isinstance(parsed, dict):
|
|
90
|
+
present = sum(1 for f in required_fields if f in parsed)
|
|
91
|
+
completeness = present / len(required_fields)
|
|
92
|
+
else:
|
|
93
|
+
completeness = 1.0
|
|
94
|
+
|
|
95
|
+
return ValidationResult(
|
|
96
|
+
valid=len(errors) == 0,
|
|
97
|
+
syntax_valid=True,
|
|
98
|
+
schema_valid=len(errors) == 0,
|
|
99
|
+
errors=errors,
|
|
100
|
+
completeness=completeness,
|
|
101
|
+
parsed=parsed,
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
def parse(self, content: str) -> Any:
|
|
105
|
+
"""Parse JSON string to Python object."""
|
|
106
|
+
return json.loads(content)
|
|
107
|
+
|
|
108
|
+
def _classify_schema_error(self, error) -> str:
|
|
109
|
+
"""Classify jsonschema error into our error types."""
|
|
110
|
+
validator = error.validator
|
|
111
|
+
if validator == "type":
|
|
112
|
+
return "type"
|
|
113
|
+
elif validator == "required":
|
|
114
|
+
return "missing"
|
|
115
|
+
elif validator == "additionalProperties":
|
|
116
|
+
return "extra"
|
|
117
|
+
elif validator in ("enum", "const"):
|
|
118
|
+
return "value"
|
|
119
|
+
elif validator in ("minLength", "maxLength", "minimum", "maximum"):
|
|
120
|
+
return "constraint"
|
|
121
|
+
else:
|
|
122
|
+
return "schema"
|
|
123
|
+
|
|
124
|
+
def _validate_schema_basic(
|
|
125
|
+
self,
|
|
126
|
+
parsed: Any,
|
|
127
|
+
schema: Dict[str, Any],
|
|
128
|
+
mode: ValidationMode,
|
|
129
|
+
) -> ValidationResult:
|
|
130
|
+
"""Basic schema validation without jsonschema library."""
|
|
131
|
+
errors = []
|
|
132
|
+
|
|
133
|
+
def check_type(value: Any, expected_type: str, path: str):
|
|
134
|
+
type_map = {
|
|
135
|
+
"string": str,
|
|
136
|
+
"number": (int, float),
|
|
137
|
+
"integer": int,
|
|
138
|
+
"boolean": bool,
|
|
139
|
+
"array": list,
|
|
140
|
+
"object": dict,
|
|
141
|
+
"null": type(None),
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
expected = type_map.get(expected_type)
|
|
145
|
+
if expected and not isinstance(value, expected):
|
|
146
|
+
if mode == ValidationMode.STRICT:
|
|
147
|
+
errors.append(ValidationError(
|
|
148
|
+
path=path,
|
|
149
|
+
message=f"Expected {expected_type}, got {type(value).__name__}",
|
|
150
|
+
error_type="type",
|
|
151
|
+
expected=expected_type,
|
|
152
|
+
actual=type(value).__name__,
|
|
153
|
+
))
|
|
154
|
+
|
|
155
|
+
def validate_object(obj: Any, obj_schema: Dict, path: str):
|
|
156
|
+
if "type" in obj_schema:
|
|
157
|
+
check_type(obj, obj_schema["type"], path)
|
|
158
|
+
|
|
159
|
+
if isinstance(obj, dict) and "properties" in obj_schema:
|
|
160
|
+
# Check required fields
|
|
161
|
+
for field in obj_schema.get("required", []):
|
|
162
|
+
if field not in obj:
|
|
163
|
+
errors.append(ValidationError(
|
|
164
|
+
path=f"{path}.{field}",
|
|
165
|
+
message=f"Missing required field: {field}",
|
|
166
|
+
error_type="missing",
|
|
167
|
+
expected=field,
|
|
168
|
+
))
|
|
169
|
+
|
|
170
|
+
# Validate properties
|
|
171
|
+
for key, prop_schema in obj_schema.get("properties", {}).items():
|
|
172
|
+
if key in obj:
|
|
173
|
+
validate_object(obj[key], prop_schema, f"{path}.{key}")
|
|
174
|
+
|
|
175
|
+
if isinstance(obj, list) and "items" in obj_schema:
|
|
176
|
+
for i, item in enumerate(obj):
|
|
177
|
+
validate_object(item, obj_schema["items"], f"{path}[{i}]")
|
|
178
|
+
|
|
179
|
+
validate_object(parsed, schema, "$")
|
|
180
|
+
|
|
181
|
+
# Calculate completeness
|
|
182
|
+
required_fields = schema.get("required", [])
|
|
183
|
+
if required_fields and isinstance(parsed, dict):
|
|
184
|
+
present = sum(1 for f in required_fields if f in parsed)
|
|
185
|
+
completeness = present / len(required_fields)
|
|
186
|
+
else:
|
|
187
|
+
completeness = 1.0
|
|
188
|
+
|
|
189
|
+
return ValidationResult(
|
|
190
|
+
valid=len(errors) == 0,
|
|
191
|
+
syntax_valid=True,
|
|
192
|
+
schema_valid=len(errors) == 0,
|
|
193
|
+
errors=errors,
|
|
194
|
+
completeness=completeness,
|
|
195
|
+
parsed=parsed,
|
|
196
|
+
)
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Pydantic model validation for LLM outputs.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from typing import Any, Dict, Optional, Type
|
|
7
|
+
from ..types import ValidationResult, ValidationError, ValidationMode
|
|
8
|
+
from .base import BaseValidator
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from pydantic import BaseModel, ValidationError as PydanticValidationError
|
|
12
|
+
_PYDANTIC_AVAILABLE = True
|
|
13
|
+
except ImportError:
|
|
14
|
+
_PYDANTIC_AVAILABLE = False
|
|
15
|
+
BaseModel = None
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class PydanticValidator(BaseValidator):
|
|
19
|
+
"""
|
|
20
|
+
Validator using Pydantic models.
|
|
21
|
+
|
|
22
|
+
Features:
|
|
23
|
+
- Full Pydantic validation with type coercion
|
|
24
|
+
- Detailed error paths
|
|
25
|
+
- Support for nested models
|
|
26
|
+
- Custom validators respected
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
format_name = "pydantic"
|
|
30
|
+
|
|
31
|
+
def __init__(self, model_class: Optional[Type] = None):
|
|
32
|
+
"""
|
|
33
|
+
Initialize with optional model class.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
model_class: Pydantic model class to validate against
|
|
37
|
+
"""
|
|
38
|
+
if not _PYDANTIC_AVAILABLE:
|
|
39
|
+
raise ImportError("Pydantic is required for PydanticValidator")
|
|
40
|
+
self.model_class = model_class
|
|
41
|
+
|
|
42
|
+
def validate_syntax(self, content: str) -> ValidationResult:
|
|
43
|
+
"""Check if content is valid JSON (required for Pydantic)."""
|
|
44
|
+
try:
|
|
45
|
+
parsed = json.loads(content)
|
|
46
|
+
return ValidationResult(
|
|
47
|
+
valid=True,
|
|
48
|
+
syntax_valid=True,
|
|
49
|
+
parsed=parsed,
|
|
50
|
+
)
|
|
51
|
+
except json.JSONDecodeError as e:
|
|
52
|
+
return ValidationResult(
|
|
53
|
+
valid=False,
|
|
54
|
+
syntax_valid=False,
|
|
55
|
+
errors=[ValidationError(
|
|
56
|
+
path=f"$.char[{e.pos}]",
|
|
57
|
+
message=f"JSON syntax error: {e.msg}",
|
|
58
|
+
error_type="syntax",
|
|
59
|
+
)]
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
def validate_schema(
|
|
63
|
+
self,
|
|
64
|
+
content: str,
|
|
65
|
+
schema: Dict[str, Any] = None,
|
|
66
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
67
|
+
) -> ValidationResult:
|
|
68
|
+
"""Validate using Pydantic model."""
|
|
69
|
+
if self.model_class is None:
|
|
70
|
+
raise ValueError("No model class set for validation")
|
|
71
|
+
|
|
72
|
+
return self.validate_model(content, self.model_class, mode)
|
|
73
|
+
|
|
74
|
+
def validate_model(
|
|
75
|
+
self,
|
|
76
|
+
content: str,
|
|
77
|
+
model_class: Type,
|
|
78
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
79
|
+
) -> ValidationResult:
|
|
80
|
+
"""
|
|
81
|
+
Validate content against a Pydantic model.
|
|
82
|
+
|
|
83
|
+
Args:
|
|
84
|
+
content: JSON string
|
|
85
|
+
model_class: Pydantic model class
|
|
86
|
+
mode: Validation mode
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
ValidationResult with Pydantic validation details
|
|
90
|
+
"""
|
|
91
|
+
# Check syntax first
|
|
92
|
+
syntax_result = self.validate_syntax(content)
|
|
93
|
+
if not syntax_result.syntax_valid:
|
|
94
|
+
return syntax_result
|
|
95
|
+
|
|
96
|
+
parsed = syntax_result.parsed
|
|
97
|
+
|
|
98
|
+
# Configure validation based on mode
|
|
99
|
+
try:
|
|
100
|
+
if mode == ValidationMode.STRICT:
|
|
101
|
+
# Pydantic v2 strict mode
|
|
102
|
+
instance = model_class.model_validate(
|
|
103
|
+
parsed,
|
|
104
|
+
strict=True,
|
|
105
|
+
)
|
|
106
|
+
else:
|
|
107
|
+
# Default: allow coercion
|
|
108
|
+
instance = model_class.model_validate(parsed)
|
|
109
|
+
|
|
110
|
+
return ValidationResult(
|
|
111
|
+
valid=True,
|
|
112
|
+
syntax_valid=True,
|
|
113
|
+
schema_valid=True,
|
|
114
|
+
completeness=1.0,
|
|
115
|
+
parsed=instance.model_dump(),
|
|
116
|
+
)
|
|
117
|
+
except PydanticValidationError as e:
|
|
118
|
+
return self._convert_pydantic_errors(e, parsed, model_class)
|
|
119
|
+
|
|
120
|
+
def _convert_pydantic_errors(
|
|
121
|
+
self,
|
|
122
|
+
pydantic_error: "PydanticValidationError",
|
|
123
|
+
parsed: Any,
|
|
124
|
+
model_class: Type,
|
|
125
|
+
) -> ValidationResult:
|
|
126
|
+
"""Convert Pydantic validation errors to our format."""
|
|
127
|
+
errors = []
|
|
128
|
+
|
|
129
|
+
for error in pydantic_error.errors():
|
|
130
|
+
# Build path from location
|
|
131
|
+
loc = error.get("loc", ())
|
|
132
|
+
path = "$" + "".join(
|
|
133
|
+
f".{p}" if isinstance(p, str) else f"[{p}]"
|
|
134
|
+
for p in loc
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
# Classify error type
|
|
138
|
+
error_type = error.get("type", "validation")
|
|
139
|
+
if "missing" in error_type:
|
|
140
|
+
classified = "missing"
|
|
141
|
+
elif "type" in error_type:
|
|
142
|
+
classified = "type"
|
|
143
|
+
elif "extra" in error_type:
|
|
144
|
+
classified = "extra"
|
|
145
|
+
else:
|
|
146
|
+
classified = "validation"
|
|
147
|
+
|
|
148
|
+
errors.append(ValidationError(
|
|
149
|
+
path=path,
|
|
150
|
+
message=error.get("msg", str(error)),
|
|
151
|
+
error_type=classified,
|
|
152
|
+
expected=error.get("ctx", {}).get("expected") if error.get("ctx") else None,
|
|
153
|
+
))
|
|
154
|
+
|
|
155
|
+
# Calculate completeness based on missing field errors
|
|
156
|
+
missing_count = sum(1 for e in errors if e.error_type == "missing")
|
|
157
|
+
total_fields = len(model_class.model_fields) if hasattr(model_class, 'model_fields') else 1
|
|
158
|
+
completeness = 1.0 - (missing_count / max(total_fields, 1))
|
|
159
|
+
|
|
160
|
+
return ValidationResult(
|
|
161
|
+
valid=False,
|
|
162
|
+
syntax_valid=True,
|
|
163
|
+
schema_valid=False,
|
|
164
|
+
errors=errors,
|
|
165
|
+
completeness=completeness,
|
|
166
|
+
parsed=parsed,
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
def parse(self, content: str) -> Any:
|
|
170
|
+
"""Parse JSON string."""
|
|
171
|
+
return json.loads(content)
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def get_schema_from_model(model_class: Type) -> Dict[str, Any]:
|
|
175
|
+
"""Extract JSON Schema from a Pydantic model."""
|
|
176
|
+
if hasattr(model_class, 'model_json_schema'):
|
|
177
|
+
return model_class.model_json_schema()
|
|
178
|
+
return {}
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""
|
|
2
|
+
YAML validation with JSON Schema support.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from typing import Any, Dict
|
|
6
|
+
from ..types import ValidationResult, ValidationError, ValidationMode
|
|
7
|
+
from .base import BaseValidator
|
|
8
|
+
|
|
9
|
+
# Optional yaml import
|
|
10
|
+
try:
|
|
11
|
+
import yaml
|
|
12
|
+
_YAML_AVAILABLE = True
|
|
13
|
+
except ImportError:
|
|
14
|
+
_YAML_AVAILABLE = False
|
|
15
|
+
|
|
16
|
+
# Optional jsonschema import
|
|
17
|
+
try:
|
|
18
|
+
import jsonschema # noqa: F401
|
|
19
|
+
from jsonschema import Draft7Validator
|
|
20
|
+
_JSONSCHEMA_AVAILABLE = True
|
|
21
|
+
except ImportError:
|
|
22
|
+
_JSONSCHEMA_AVAILABLE = False
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class YAMLValidator(BaseValidator):
|
|
26
|
+
"""
|
|
27
|
+
Validator for YAML output.
|
|
28
|
+
|
|
29
|
+
Features:
|
|
30
|
+
- Syntax validation (yaml.safe_load)
|
|
31
|
+
- JSON Schema validation (YAML is a superset of JSON)
|
|
32
|
+
- Support for multi-document YAML
|
|
33
|
+
- Handles common LLM YAML mistakes
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
format_name = "yaml"
|
|
37
|
+
|
|
38
|
+
def __init__(self, allow_multi_doc: bool = False):
|
|
39
|
+
"""
|
|
40
|
+
Initialize YAML validator.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
allow_multi_doc: Allow multiple YAML documents (---) in content
|
|
44
|
+
"""
|
|
45
|
+
if not _YAML_AVAILABLE:
|
|
46
|
+
raise ImportError("PyYAML is required for YAMLValidator")
|
|
47
|
+
self.allow_multi_doc = allow_multi_doc
|
|
48
|
+
|
|
49
|
+
def validate_syntax(self, content: str) -> ValidationResult:
|
|
50
|
+
"""Check if content is valid YAML."""
|
|
51
|
+
try:
|
|
52
|
+
# Try to fix common LLM YAML mistakes
|
|
53
|
+
content = self._preprocess_yaml(content)
|
|
54
|
+
|
|
55
|
+
if self.allow_multi_doc:
|
|
56
|
+
parsed = list(yaml.safe_load_all(content))
|
|
57
|
+
if len(parsed) == 1:
|
|
58
|
+
parsed = parsed[0]
|
|
59
|
+
else:
|
|
60
|
+
parsed = yaml.safe_load(content)
|
|
61
|
+
|
|
62
|
+
return ValidationResult(
|
|
63
|
+
valid=True,
|
|
64
|
+
syntax_valid=True,
|
|
65
|
+
parsed=parsed,
|
|
66
|
+
)
|
|
67
|
+
except yaml.YAMLError as e:
|
|
68
|
+
error_msg = str(e)
|
|
69
|
+
line = getattr(e, 'problem_mark', None)
|
|
70
|
+
if line:
|
|
71
|
+
path = f"$.line[{line.line + 1}]:col[{line.column + 1}]"
|
|
72
|
+
else:
|
|
73
|
+
path = "$"
|
|
74
|
+
|
|
75
|
+
return ValidationResult(
|
|
76
|
+
valid=False,
|
|
77
|
+
syntax_valid=False,
|
|
78
|
+
errors=[ValidationError(
|
|
79
|
+
path=path,
|
|
80
|
+
message=f"YAML syntax error: {error_msg}",
|
|
81
|
+
error_type="syntax",
|
|
82
|
+
)]
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
def validate_schema(
|
|
86
|
+
self,
|
|
87
|
+
content: str,
|
|
88
|
+
schema: Dict[str, Any],
|
|
89
|
+
mode: ValidationMode = ValidationMode.COERCE,
|
|
90
|
+
) -> ValidationResult:
|
|
91
|
+
"""Validate YAML against JSON Schema."""
|
|
92
|
+
# First check syntax
|
|
93
|
+
syntax_result = self.validate_syntax(content)
|
|
94
|
+
if not syntax_result.syntax_valid:
|
|
95
|
+
return syntax_result
|
|
96
|
+
|
|
97
|
+
parsed = syntax_result.parsed
|
|
98
|
+
|
|
99
|
+
if not _JSONSCHEMA_AVAILABLE:
|
|
100
|
+
# Fallback to basic type checking
|
|
101
|
+
return self._validate_schema_basic(parsed, schema, mode)
|
|
102
|
+
|
|
103
|
+
# Use jsonschema for validation (YAML parses to same types as JSON)
|
|
104
|
+
errors = []
|
|
105
|
+
validator = Draft7Validator(schema)
|
|
106
|
+
|
|
107
|
+
for error in validator.iter_errors(parsed):
|
|
108
|
+
path = "$" + "".join(
|
|
109
|
+
f".{p}" if isinstance(p, str) else f"[{p}]"
|
|
110
|
+
for p in error.absolute_path
|
|
111
|
+
)
|
|
112
|
+
errors.append(ValidationError(
|
|
113
|
+
path=path,
|
|
114
|
+
message=error.message,
|
|
115
|
+
error_type=self._classify_schema_error(error),
|
|
116
|
+
expected=error.schema.get("type") if hasattr(error, 'schema') else None,
|
|
117
|
+
actual=type(error.instance).__name__ if error.instance is not None else None,
|
|
118
|
+
))
|
|
119
|
+
|
|
120
|
+
# Calculate completeness
|
|
121
|
+
required_fields = schema.get("required", [])
|
|
122
|
+
if required_fields and isinstance(parsed, dict):
|
|
123
|
+
present = sum(1 for f in required_fields if f in parsed)
|
|
124
|
+
completeness = present / len(required_fields)
|
|
125
|
+
else:
|
|
126
|
+
completeness = 1.0
|
|
127
|
+
|
|
128
|
+
return ValidationResult(
|
|
129
|
+
valid=len(errors) == 0,
|
|
130
|
+
syntax_valid=True,
|
|
131
|
+
schema_valid=len(errors) == 0,
|
|
132
|
+
errors=errors,
|
|
133
|
+
completeness=completeness,
|
|
134
|
+
parsed=parsed,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def parse(self, content: str) -> Any:
|
|
138
|
+
"""Parse YAML string to Python object."""
|
|
139
|
+
content = self._preprocess_yaml(content)
|
|
140
|
+
return yaml.safe_load(content)
|
|
141
|
+
|
|
142
|
+
def _preprocess_yaml(self, content: str) -> str:
|
|
143
|
+
"""Fix common LLM YAML mistakes."""
|
|
144
|
+
lines = content.split('\n')
|
|
145
|
+
fixed_lines = []
|
|
146
|
+
|
|
147
|
+
for line in lines:
|
|
148
|
+
# Fix tabs (YAML doesn't allow tabs for indentation)
|
|
149
|
+
if '\t' in line:
|
|
150
|
+
# Replace tabs with 2 spaces
|
|
151
|
+
line = line.replace('\t', ' ')
|
|
152
|
+
|
|
153
|
+
# Remove trailing whitespace
|
|
154
|
+
line = line.rstrip()
|
|
155
|
+
|
|
156
|
+
fixed_lines.append(line)
|
|
157
|
+
|
|
158
|
+
return '\n'.join(fixed_lines)
|
|
159
|
+
|
|
160
|
+
def _classify_schema_error(self, error) -> str:
|
|
161
|
+
"""Classify jsonschema error into our error types."""
|
|
162
|
+
validator = error.validator
|
|
163
|
+
if validator == "type":
|
|
164
|
+
return "type"
|
|
165
|
+
elif validator == "required":
|
|
166
|
+
return "missing"
|
|
167
|
+
elif validator == "additionalProperties":
|
|
168
|
+
return "extra"
|
|
169
|
+
elif validator in ("enum", "const"):
|
|
170
|
+
return "value"
|
|
171
|
+
elif validator in ("minLength", "maxLength", "minimum", "maximum"):
|
|
172
|
+
return "constraint"
|
|
173
|
+
else:
|
|
174
|
+
return "schema"
|
|
175
|
+
|
|
176
|
+
def _validate_schema_basic(
|
|
177
|
+
self,
|
|
178
|
+
parsed: Any,
|
|
179
|
+
schema: Dict[str, Any],
|
|
180
|
+
mode: ValidationMode,
|
|
181
|
+
) -> ValidationResult:
|
|
182
|
+
"""Basic schema validation without jsonschema library."""
|
|
183
|
+
errors = []
|
|
184
|
+
|
|
185
|
+
def check_type(value: Any, expected_type: str, path: str):
|
|
186
|
+
type_map = {
|
|
187
|
+
"string": str,
|
|
188
|
+
"number": (int, float),
|
|
189
|
+
"integer": int,
|
|
190
|
+
"boolean": bool,
|
|
191
|
+
"array": list,
|
|
192
|
+
"object": dict,
|
|
193
|
+
"null": type(None),
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
expected = type_map.get(expected_type)
|
|
197
|
+
if expected and not isinstance(value, expected):
|
|
198
|
+
if mode == ValidationMode.STRICT:
|
|
199
|
+
errors.append(ValidationError(
|
|
200
|
+
path=path,
|
|
201
|
+
message=f"Expected {expected_type}, got {type(value).__name__}",
|
|
202
|
+
error_type="type",
|
|
203
|
+
expected=expected_type,
|
|
204
|
+
actual=type(value).__name__,
|
|
205
|
+
))
|
|
206
|
+
|
|
207
|
+
def validate_object(obj: Any, obj_schema: Dict, path: str):
|
|
208
|
+
if "type" in obj_schema:
|
|
209
|
+
check_type(obj, obj_schema["type"], path)
|
|
210
|
+
|
|
211
|
+
if isinstance(obj, dict) and "properties" in obj_schema:
|
|
212
|
+
# Check required fields
|
|
213
|
+
for field in obj_schema.get("required", []):
|
|
214
|
+
if field not in obj:
|
|
215
|
+
errors.append(ValidationError(
|
|
216
|
+
path=f"{path}.{field}",
|
|
217
|
+
message=f"Missing required field: {field}",
|
|
218
|
+
error_type="missing",
|
|
219
|
+
expected=field,
|
|
220
|
+
))
|
|
221
|
+
|
|
222
|
+
# Validate properties
|
|
223
|
+
for key, prop_schema in obj_schema.get("properties", {}).items():
|
|
224
|
+
if key in obj:
|
|
225
|
+
validate_object(obj[key], prop_schema, f"{path}.{key}")
|
|
226
|
+
|
|
227
|
+
if isinstance(obj, list) and "items" in obj_schema:
|
|
228
|
+
for i, item in enumerate(obj):
|
|
229
|
+
validate_object(item, obj_schema["items"], f"{path}[{i}]")
|
|
230
|
+
|
|
231
|
+
validate_object(parsed, schema, "$")
|
|
232
|
+
|
|
233
|
+
# Calculate completeness
|
|
234
|
+
required_fields = schema.get("required", [])
|
|
235
|
+
if required_fields and isinstance(parsed, dict):
|
|
236
|
+
present = sum(1 for f in required_fields if f in parsed)
|
|
237
|
+
completeness = present / len(required_fields)
|
|
238
|
+
else:
|
|
239
|
+
completeness = 1.0
|
|
240
|
+
|
|
241
|
+
return ValidationResult(
|
|
242
|
+
valid=len(errors) == 0,
|
|
243
|
+
syntax_valid=True,
|
|
244
|
+
schema_valid=len(errors) == 0,
|
|
245
|
+
errors=errors,
|
|
246
|
+
completeness=completeness,
|
|
247
|
+
parsed=parsed,
|
|
248
|
+
)
|