agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,551 @@
|
|
|
1
|
+
"""Streaming Evaluator for real-time LLM output evaluation.
|
|
2
|
+
|
|
3
|
+
Evaluates LLM outputs in real-time as tokens stream in, with support for
|
|
4
|
+
early stopping based on configurable policies.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import asyncio
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from typing import (
|
|
11
|
+
AsyncIterator,
|
|
12
|
+
Callable,
|
|
13
|
+
Dict,
|
|
14
|
+
Iterator,
|
|
15
|
+
List,
|
|
16
|
+
Optional,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
from .types import (
|
|
20
|
+
ChunkResult,
|
|
21
|
+
EarlyStopReason,
|
|
22
|
+
StreamingConfig,
|
|
23
|
+
StreamingEvalResult,
|
|
24
|
+
StreamingState,
|
|
25
|
+
)
|
|
26
|
+
from .buffer import ChunkBuffer
|
|
27
|
+
from .policy import EarlyStopPolicy
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class EvalSpec:
|
|
32
|
+
"""Specification for a streaming evaluation."""
|
|
33
|
+
|
|
34
|
+
name: str
|
|
35
|
+
eval_fn: Callable[[str, str], float] # (chunk, cumulative) -> score
|
|
36
|
+
threshold: float = 0.7
|
|
37
|
+
weight: float = 1.0
|
|
38
|
+
pass_above: bool = True # True if higher scores are better
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class StreamingEvaluator:
|
|
42
|
+
"""
|
|
43
|
+
Evaluates LLM outputs in real-time as tokens stream in.
|
|
44
|
+
|
|
45
|
+
Supports early stopping based on configurable policies and provides
|
|
46
|
+
detailed per-chunk and aggregate evaluation results.
|
|
47
|
+
|
|
48
|
+
Example:
|
|
49
|
+
evaluator = StreamingEvaluator(config)
|
|
50
|
+
evaluator.add_eval("toxicity", toxicity_scorer, threshold=0.7, pass_above=False)
|
|
51
|
+
evaluator.set_policy(EarlyStopPolicy.default())
|
|
52
|
+
|
|
53
|
+
# Synchronous iteration
|
|
54
|
+
for token in stream:
|
|
55
|
+
result = evaluator.process_token(token)
|
|
56
|
+
if result and result.should_stop:
|
|
57
|
+
break
|
|
58
|
+
|
|
59
|
+
final_result = evaluator.finalize()
|
|
60
|
+
|
|
61
|
+
# Or async iteration
|
|
62
|
+
async for token in async_stream:
|
|
63
|
+
result = await evaluator.process_token_async(token)
|
|
64
|
+
if result and result.should_stop:
|
|
65
|
+
break
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(
|
|
69
|
+
self,
|
|
70
|
+
config: Optional[StreamingConfig] = None,
|
|
71
|
+
policy: Optional[EarlyStopPolicy] = None,
|
|
72
|
+
):
|
|
73
|
+
"""
|
|
74
|
+
Initialize the streaming evaluator.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
config: Streaming configuration (uses defaults if None)
|
|
78
|
+
policy: Early stop policy (uses default if None)
|
|
79
|
+
"""
|
|
80
|
+
self.config = config or StreamingConfig()
|
|
81
|
+
self._policy = policy or EarlyStopPolicy.default()
|
|
82
|
+
self._buffer = ChunkBuffer(self.config)
|
|
83
|
+
self._evals: List[EvalSpec] = []
|
|
84
|
+
self._chunk_results: List[ChunkResult] = []
|
|
85
|
+
self._state = StreamingState.IDLE
|
|
86
|
+
self._start_time: float = 0.0
|
|
87
|
+
self._stop_reason = EarlyStopReason.NONE
|
|
88
|
+
self._stopped_at_chunk: Optional[int] = None
|
|
89
|
+
|
|
90
|
+
def add_eval(
|
|
91
|
+
self,
|
|
92
|
+
name: str,
|
|
93
|
+
eval_fn: Callable[[str, str], float],
|
|
94
|
+
threshold: float = 0.7,
|
|
95
|
+
weight: float = 1.0,
|
|
96
|
+
pass_above: bool = True,
|
|
97
|
+
) -> "StreamingEvaluator":
|
|
98
|
+
"""
|
|
99
|
+
Add an evaluation function.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
name: Name of the evaluation
|
|
103
|
+
eval_fn: Function that takes (chunk_text, cumulative_text) and returns score
|
|
104
|
+
threshold: Passing threshold
|
|
105
|
+
weight: Weight for final score calculation
|
|
106
|
+
pass_above: If True, scores above threshold pass; if False, below
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
Self for chaining
|
|
110
|
+
"""
|
|
111
|
+
self._evals.append(
|
|
112
|
+
EvalSpec(
|
|
113
|
+
name=name,
|
|
114
|
+
eval_fn=eval_fn,
|
|
115
|
+
threshold=threshold,
|
|
116
|
+
weight=weight,
|
|
117
|
+
pass_above=pass_above,
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
return self
|
|
121
|
+
|
|
122
|
+
def set_policy(self, policy: EarlyStopPolicy) -> "StreamingEvaluator":
|
|
123
|
+
"""
|
|
124
|
+
Set the early stop policy.
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
policy: Policy to use for early stopping
|
|
128
|
+
|
|
129
|
+
Returns:
|
|
130
|
+
Self for chaining
|
|
131
|
+
"""
|
|
132
|
+
self._policy = policy
|
|
133
|
+
return self
|
|
134
|
+
|
|
135
|
+
def reset(self) -> None:
|
|
136
|
+
"""Reset the evaluator for a new stream."""
|
|
137
|
+
self._buffer.reset()
|
|
138
|
+
self._policy.reset()
|
|
139
|
+
self._chunk_results = []
|
|
140
|
+
self._state = StreamingState.IDLE
|
|
141
|
+
self._start_time = 0.0
|
|
142
|
+
self._stop_reason = EarlyStopReason.NONE
|
|
143
|
+
self._stopped_at_chunk = None
|
|
144
|
+
|
|
145
|
+
def process_token(self, token: str) -> Optional[ChunkResult]:
|
|
146
|
+
"""
|
|
147
|
+
Process a single token from the stream.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
token: The token text
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
ChunkResult if evaluation was triggered, None otherwise
|
|
154
|
+
"""
|
|
155
|
+
if self._state == StreamingState.IDLE:
|
|
156
|
+
self._state = StreamingState.STREAMING
|
|
157
|
+
self._start_time = time.perf_counter()
|
|
158
|
+
|
|
159
|
+
if self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR):
|
|
160
|
+
return None
|
|
161
|
+
|
|
162
|
+
# Add token to buffer
|
|
163
|
+
self._buffer.add(token)
|
|
164
|
+
|
|
165
|
+
# Check for limits
|
|
166
|
+
should_stop_limits, limit_reason = self._buffer.should_stop_for_limits()
|
|
167
|
+
if should_stop_limits:
|
|
168
|
+
return self._handle_limit_stop(limit_reason)
|
|
169
|
+
|
|
170
|
+
# Check if we should evaluate
|
|
171
|
+
if not self._buffer.should_evaluate():
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
# Run evaluation
|
|
175
|
+
return self._evaluate_chunk()
|
|
176
|
+
|
|
177
|
+
async def process_token_async(self, token: str) -> Optional[ChunkResult]:
|
|
178
|
+
"""
|
|
179
|
+
Process a single token asynchronously.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
token: The token text
|
|
183
|
+
|
|
184
|
+
Returns:
|
|
185
|
+
ChunkResult if evaluation was triggered, None otherwise
|
|
186
|
+
"""
|
|
187
|
+
# For now, wrap sync processing - can be optimized later
|
|
188
|
+
return await asyncio.to_thread(self.process_token, token)
|
|
189
|
+
|
|
190
|
+
def process_chunk(self, chunk: str) -> Optional[ChunkResult]:
|
|
191
|
+
"""
|
|
192
|
+
Process a larger chunk of text (multiple tokens).
|
|
193
|
+
|
|
194
|
+
Args:
|
|
195
|
+
chunk: The chunk text
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
ChunkResult if evaluation was triggered, None otherwise
|
|
199
|
+
"""
|
|
200
|
+
if self._state == StreamingState.IDLE:
|
|
201
|
+
self._state = StreamingState.STREAMING
|
|
202
|
+
self._start_time = time.perf_counter()
|
|
203
|
+
|
|
204
|
+
if self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR):
|
|
205
|
+
return None
|
|
206
|
+
|
|
207
|
+
# Add chunk to buffer
|
|
208
|
+
self._buffer.add_chunk(chunk)
|
|
209
|
+
|
|
210
|
+
# Check for limits
|
|
211
|
+
should_stop_limits, limit_reason = self._buffer.should_stop_for_limits()
|
|
212
|
+
if should_stop_limits:
|
|
213
|
+
return self._handle_limit_stop(limit_reason)
|
|
214
|
+
|
|
215
|
+
# Check if we should evaluate
|
|
216
|
+
if not self._buffer.should_evaluate():
|
|
217
|
+
return None
|
|
218
|
+
|
|
219
|
+
# Run evaluation
|
|
220
|
+
return self._evaluate_chunk()
|
|
221
|
+
|
|
222
|
+
def _evaluate_chunk(self) -> ChunkResult:
|
|
223
|
+
"""Run evaluations on the current chunk."""
|
|
224
|
+
chunk_start = time.perf_counter()
|
|
225
|
+
|
|
226
|
+
chunk_text = self._buffer.get_chunk()
|
|
227
|
+
cumulative_text = self._buffer.get_cumulative()
|
|
228
|
+
chunk_index = self._buffer.get_chunk_index()
|
|
229
|
+
|
|
230
|
+
scores: Dict[str, float] = {}
|
|
231
|
+
flags: Dict[str, bool] = {}
|
|
232
|
+
|
|
233
|
+
# Run all evaluations
|
|
234
|
+
for eval_spec in self._evals:
|
|
235
|
+
try:
|
|
236
|
+
score = eval_spec.eval_fn(chunk_text, cumulative_text)
|
|
237
|
+
scores[eval_spec.name] = score
|
|
238
|
+
|
|
239
|
+
# Determine if passed
|
|
240
|
+
if eval_spec.pass_above:
|
|
241
|
+
flags[eval_spec.name] = score >= eval_spec.threshold
|
|
242
|
+
else:
|
|
243
|
+
flags[eval_spec.name] = score <= eval_spec.threshold
|
|
244
|
+
except Exception:
|
|
245
|
+
# Handle eval errors gracefully
|
|
246
|
+
scores[eval_spec.name] = 0.0
|
|
247
|
+
flags[eval_spec.name] = False
|
|
248
|
+
|
|
249
|
+
latency_ms = (time.perf_counter() - chunk_start) * 1000
|
|
250
|
+
|
|
251
|
+
# Create chunk result
|
|
252
|
+
chunk_result = ChunkResult(
|
|
253
|
+
chunk_index=chunk_index,
|
|
254
|
+
chunk_text=chunk_text,
|
|
255
|
+
cumulative_text=cumulative_text,
|
|
256
|
+
scores=scores,
|
|
257
|
+
flags=flags,
|
|
258
|
+
latency_ms=latency_ms,
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
# Check policy for early stop
|
|
262
|
+
if self.config.enable_early_stop:
|
|
263
|
+
should_stop, stop_reason = self._policy.check(chunk_result)
|
|
264
|
+
|
|
265
|
+
if should_stop:
|
|
266
|
+
chunk_result.should_stop = True
|
|
267
|
+
chunk_result.stop_reason = stop_reason
|
|
268
|
+
self._state = StreamingState.STOPPED
|
|
269
|
+
self._stop_reason = stop_reason
|
|
270
|
+
self._stopped_at_chunk = chunk_index
|
|
271
|
+
|
|
272
|
+
# Trigger callback if configured
|
|
273
|
+
if self.config.on_stop_callback:
|
|
274
|
+
self.config.on_stop_callback(stop_reason, cumulative_text)
|
|
275
|
+
|
|
276
|
+
# Check stop on first failure
|
|
277
|
+
elif self.config.stop_on_first_failure and not chunk_result.all_passed:
|
|
278
|
+
chunk_result.should_stop = True
|
|
279
|
+
chunk_result.stop_reason = EarlyStopReason.THRESHOLD
|
|
280
|
+
self._state = StreamingState.STOPPED
|
|
281
|
+
self._stop_reason = EarlyStopReason.THRESHOLD
|
|
282
|
+
self._stopped_at_chunk = chunk_index
|
|
283
|
+
|
|
284
|
+
# Trigger callback if configured
|
|
285
|
+
if self.config.on_stop_callback:
|
|
286
|
+
self.config.on_stop_callback(EarlyStopReason.THRESHOLD, cumulative_text)
|
|
287
|
+
|
|
288
|
+
# Store result
|
|
289
|
+
self._chunk_results.append(chunk_result)
|
|
290
|
+
|
|
291
|
+
# Mark as evaluated
|
|
292
|
+
self._buffer.mark_evaluated()
|
|
293
|
+
|
|
294
|
+
# Trigger chunk callback if configured
|
|
295
|
+
if self.config.on_chunk_callback:
|
|
296
|
+
self.config.on_chunk_callback(chunk_result)
|
|
297
|
+
|
|
298
|
+
return chunk_result
|
|
299
|
+
|
|
300
|
+
def _handle_limit_stop(self, reason: str) -> ChunkResult:
|
|
301
|
+
"""Handle stopping due to limits."""
|
|
302
|
+
# Evaluate any remaining content first
|
|
303
|
+
if self._buffer.has_pending:
|
|
304
|
+
chunk_result = self._evaluate_chunk()
|
|
305
|
+
else:
|
|
306
|
+
# Create a minimal result for the stop
|
|
307
|
+
chunk_result = ChunkResult(
|
|
308
|
+
chunk_index=self._buffer.get_chunk_index(),
|
|
309
|
+
chunk_text="",
|
|
310
|
+
cumulative_text=self._buffer.get_cumulative(),
|
|
311
|
+
scores={},
|
|
312
|
+
flags={},
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
# Map limit reason to stop reason
|
|
316
|
+
if reason == "max_tokens":
|
|
317
|
+
stop_reason = EarlyStopReason.MAX_TOKENS
|
|
318
|
+
elif reason == "max_chars":
|
|
319
|
+
stop_reason = EarlyStopReason.MAX_CHARS
|
|
320
|
+
elif reason == "timeout":
|
|
321
|
+
stop_reason = EarlyStopReason.TIMEOUT
|
|
322
|
+
else:
|
|
323
|
+
stop_reason = EarlyStopReason.ERROR
|
|
324
|
+
|
|
325
|
+
chunk_result.should_stop = True
|
|
326
|
+
chunk_result.stop_reason = stop_reason
|
|
327
|
+
self._state = StreamingState.STOPPED
|
|
328
|
+
self._stop_reason = stop_reason
|
|
329
|
+
self._stopped_at_chunk = chunk_result.chunk_index
|
|
330
|
+
|
|
331
|
+
return chunk_result
|
|
332
|
+
|
|
333
|
+
def finalize(self) -> StreamingEvalResult:
|
|
334
|
+
"""
|
|
335
|
+
Finalize evaluation and return results.
|
|
336
|
+
|
|
337
|
+
Should be called after stream completes or after early stop.
|
|
338
|
+
|
|
339
|
+
Returns:
|
|
340
|
+
StreamingEvalResult with all evaluation data
|
|
341
|
+
"""
|
|
342
|
+
# Evaluate any remaining pending content
|
|
343
|
+
if self._buffer.has_pending and self._state == StreamingState.STREAMING:
|
|
344
|
+
self._evaluate_chunk()
|
|
345
|
+
|
|
346
|
+
# Mark as completed if not already stopped
|
|
347
|
+
if self._state == StreamingState.STREAMING:
|
|
348
|
+
self._state = StreamingState.COMPLETED
|
|
349
|
+
|
|
350
|
+
# Calculate final scores (weighted average across chunks)
|
|
351
|
+
final_scores = self._calculate_final_scores()
|
|
352
|
+
|
|
353
|
+
# Determine overall pass/fail
|
|
354
|
+
passed = self._determine_passed(final_scores)
|
|
355
|
+
|
|
356
|
+
# Calculate total latency
|
|
357
|
+
total_latency_ms = (time.perf_counter() - self._start_time) * 1000 if self._start_time else 0.0
|
|
358
|
+
|
|
359
|
+
return StreamingEvalResult(
|
|
360
|
+
passed=passed,
|
|
361
|
+
final_text=self._buffer.get_cumulative(),
|
|
362
|
+
total_chunks=len(self._chunk_results),
|
|
363
|
+
chunk_results=self._chunk_results,
|
|
364
|
+
final_scores=final_scores,
|
|
365
|
+
early_stopped=self._state == StreamingState.STOPPED,
|
|
366
|
+
stop_reason=self._stop_reason,
|
|
367
|
+
stopped_at_chunk=self._stopped_at_chunk,
|
|
368
|
+
total_latency_ms=total_latency_ms,
|
|
369
|
+
state=self._state,
|
|
370
|
+
metadata={
|
|
371
|
+
"buffer_stats": self._buffer.get_stats(),
|
|
372
|
+
"policy_stats": self._policy.get_stats(),
|
|
373
|
+
},
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
def _calculate_final_scores(self) -> Dict[str, float]:
|
|
377
|
+
"""Calculate final weighted average scores."""
|
|
378
|
+
if not self._chunk_results:
|
|
379
|
+
return {}
|
|
380
|
+
|
|
381
|
+
# Build weight lookup from eval specs
|
|
382
|
+
weights: Dict[str, float] = {e.name: e.weight for e in self._evals}
|
|
383
|
+
|
|
384
|
+
final_scores: Dict[str, float] = {}
|
|
385
|
+
weighted_sums: Dict[str, float] = {}
|
|
386
|
+
weight_totals: Dict[str, float] = {}
|
|
387
|
+
|
|
388
|
+
for chunk_result in self._chunk_results:
|
|
389
|
+
for name, score in chunk_result.scores.items():
|
|
390
|
+
w = weights.get(name, 1.0)
|
|
391
|
+
if name not in weighted_sums:
|
|
392
|
+
weighted_sums[name] = 0.0
|
|
393
|
+
weight_totals[name] = 0.0
|
|
394
|
+
weighted_sums[name] += score * w
|
|
395
|
+
weight_totals[name] += w
|
|
396
|
+
|
|
397
|
+
for name in weighted_sums:
|
|
398
|
+
if weight_totals[name] > 0:
|
|
399
|
+
final_scores[name] = weighted_sums[name] / weight_totals[name]
|
|
400
|
+
else:
|
|
401
|
+
final_scores[name] = 0.0
|
|
402
|
+
|
|
403
|
+
return final_scores
|
|
404
|
+
|
|
405
|
+
def _determine_passed(self, final_scores: Dict[str, float]) -> bool:
|
|
406
|
+
"""Determine if evaluation passed overall."""
|
|
407
|
+
if self._state == StreamingState.STOPPED:
|
|
408
|
+
# If stopped early due to safety/toxicity, fail
|
|
409
|
+
if self._stop_reason in (
|
|
410
|
+
EarlyStopReason.TOXICITY,
|
|
411
|
+
EarlyStopReason.SAFETY,
|
|
412
|
+
EarlyStopReason.PII,
|
|
413
|
+
EarlyStopReason.JAILBREAK,
|
|
414
|
+
):
|
|
415
|
+
return False
|
|
416
|
+
|
|
417
|
+
# Check final scores against thresholds
|
|
418
|
+
for eval_spec in self._evals:
|
|
419
|
+
if eval_spec.name in final_scores:
|
|
420
|
+
score = final_scores[eval_spec.name]
|
|
421
|
+
if eval_spec.pass_above:
|
|
422
|
+
if score < eval_spec.threshold:
|
|
423
|
+
return False
|
|
424
|
+
else:
|
|
425
|
+
if score > eval_spec.threshold:
|
|
426
|
+
return False
|
|
427
|
+
|
|
428
|
+
return True
|
|
429
|
+
|
|
430
|
+
@property
|
|
431
|
+
def state(self) -> StreamingState:
|
|
432
|
+
"""Get current evaluation state."""
|
|
433
|
+
return self._state
|
|
434
|
+
|
|
435
|
+
@property
|
|
436
|
+
def is_stopped(self) -> bool:
|
|
437
|
+
"""Check if evaluation has stopped."""
|
|
438
|
+
return self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR)
|
|
439
|
+
|
|
440
|
+
@property
|
|
441
|
+
def chunk_count(self) -> int:
|
|
442
|
+
"""Get number of chunks evaluated."""
|
|
443
|
+
return len(self._chunk_results)
|
|
444
|
+
|
|
445
|
+
def evaluate_stream(
|
|
446
|
+
self,
|
|
447
|
+
stream: Iterator[str],
|
|
448
|
+
) -> StreamingEvalResult:
|
|
449
|
+
"""
|
|
450
|
+
Evaluate an entire stream synchronously.
|
|
451
|
+
|
|
452
|
+
Args:
|
|
453
|
+
stream: Iterator yielding tokens/chunks
|
|
454
|
+
|
|
455
|
+
Returns:
|
|
456
|
+
StreamingEvalResult after processing complete stream
|
|
457
|
+
"""
|
|
458
|
+
self.reset()
|
|
459
|
+
|
|
460
|
+
for token in stream:
|
|
461
|
+
result = self.process_token(token)
|
|
462
|
+
if result and result.should_stop:
|
|
463
|
+
break
|
|
464
|
+
|
|
465
|
+
return self.finalize()
|
|
466
|
+
|
|
467
|
+
async def evaluate_stream_async(
|
|
468
|
+
self,
|
|
469
|
+
stream: AsyncIterator[str],
|
|
470
|
+
) -> StreamingEvalResult:
|
|
471
|
+
"""
|
|
472
|
+
Evaluate an entire stream asynchronously.
|
|
473
|
+
|
|
474
|
+
Args:
|
|
475
|
+
stream: Async iterator yielding tokens/chunks
|
|
476
|
+
|
|
477
|
+
Returns:
|
|
478
|
+
StreamingEvalResult after processing complete stream
|
|
479
|
+
"""
|
|
480
|
+
self.reset()
|
|
481
|
+
|
|
482
|
+
async for token in stream:
|
|
483
|
+
result = await self.process_token_async(token)
|
|
484
|
+
if result and result.should_stop:
|
|
485
|
+
break
|
|
486
|
+
|
|
487
|
+
return self.finalize()
|
|
488
|
+
|
|
489
|
+
@classmethod
|
|
490
|
+
def with_defaults(cls) -> "StreamingEvaluator":
|
|
491
|
+
"""
|
|
492
|
+
Create an evaluator with default configuration.
|
|
493
|
+
|
|
494
|
+
Returns:
|
|
495
|
+
StreamingEvaluator with default settings
|
|
496
|
+
"""
|
|
497
|
+
return cls(
|
|
498
|
+
config=StreamingConfig(),
|
|
499
|
+
policy=EarlyStopPolicy.default(),
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
@classmethod
|
|
503
|
+
def for_safety(
|
|
504
|
+
cls,
|
|
505
|
+
toxicity_threshold: float = 0.5,
|
|
506
|
+
safety_threshold: float = 0.5,
|
|
507
|
+
) -> "StreamingEvaluator":
|
|
508
|
+
"""
|
|
509
|
+
Create an evaluator optimized for safety monitoring.
|
|
510
|
+
|
|
511
|
+
Args:
|
|
512
|
+
toxicity_threshold: Threshold for toxicity (stop if above)
|
|
513
|
+
safety_threshold: Threshold for safety (stop if below)
|
|
514
|
+
|
|
515
|
+
Returns:
|
|
516
|
+
StreamingEvaluator configured for safety
|
|
517
|
+
"""
|
|
518
|
+
config = StreamingConfig(
|
|
519
|
+
enable_early_stop=True,
|
|
520
|
+
stop_on_first_failure=True,
|
|
521
|
+
toxicity_threshold=toxicity_threshold,
|
|
522
|
+
safety_threshold=safety_threshold,
|
|
523
|
+
)
|
|
524
|
+
policy = EarlyStopPolicy.strict()
|
|
525
|
+
return cls(config=config, policy=policy)
|
|
526
|
+
|
|
527
|
+
@classmethod
|
|
528
|
+
def for_quality(
|
|
529
|
+
cls,
|
|
530
|
+
min_chunk_size: int = 50,
|
|
531
|
+
eval_interval_ms: int = 500,
|
|
532
|
+
) -> "StreamingEvaluator":
|
|
533
|
+
"""
|
|
534
|
+
Create an evaluator optimized for quality assessment.
|
|
535
|
+
|
|
536
|
+
Args:
|
|
537
|
+
min_chunk_size: Minimum characters before evaluation
|
|
538
|
+
eval_interval_ms: Milliseconds between evaluations
|
|
539
|
+
|
|
540
|
+
Returns:
|
|
541
|
+
StreamingEvaluator configured for quality
|
|
542
|
+
"""
|
|
543
|
+
config = StreamingConfig(
|
|
544
|
+
min_chunk_size=min_chunk_size,
|
|
545
|
+
max_chunk_size=200,
|
|
546
|
+
eval_interval_ms=eval_interval_ms,
|
|
547
|
+
enable_early_stop=False, # Don't stop early for quality
|
|
548
|
+
eval_on_sentence_end=True,
|
|
549
|
+
)
|
|
550
|
+
policy = EarlyStopPolicy.permissive()
|
|
551
|
+
return cls(config=config, policy=policy)
|