agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
import base64
|
|
2
|
+
import os
|
|
3
|
+
from typing import ClassVar, Set
|
|
4
|
+
from urllib.parse import urlparse
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, field_validator
|
|
7
|
+
|
|
8
|
+
# Keep this class minimal & fast: no network calls; only local file -> data URI.
|
|
9
|
+
# The backend will detect input type and do any heavy lifting.
|
|
10
|
+
|
|
11
|
+
class ProtectInputAdapter(BaseModel):
|
|
12
|
+
"""
|
|
13
|
+
Minimal, production-safe input wrapper for Protect.
|
|
14
|
+
|
|
15
|
+
- Accepts text, http(s) URLs, data URIs, and local audio/image files.
|
|
16
|
+
- Rejects known HTML 'viewer' URLs (e.g., GitHub blob) to avoid 400s downstream.
|
|
17
|
+
- Converts local media files to 'data:' URIs (cheap; no networking).
|
|
18
|
+
- Leaves type detection to the backend.
|
|
19
|
+
"""
|
|
20
|
+
input: str
|
|
21
|
+
call_type: str = "protect"
|
|
22
|
+
|
|
23
|
+
AUDIO_EXTENSIONS: ClassVar[Set[str]] = {
|
|
24
|
+
".mp3", ".wav",
|
|
25
|
+
}
|
|
26
|
+
IMAGE_EXTENSIONS: ClassVar[Set[str]] = {
|
|
27
|
+
".jpg", ".jpeg", ".png", ".gif", ".webp", ".bmp", ".tiff", ".tif", ".svg"
|
|
28
|
+
}
|
|
29
|
+
# Optional cap for local files (bytes). Keep small to protect latency/footguns.
|
|
30
|
+
MAX_LOCAL_BYTES: ClassVar[int] = int(os.getenv("FI_PROTECT_MAX_LOCAL_BYTES", "20000000")) # 20MB
|
|
31
|
+
|
|
32
|
+
# --- Pydantic v2: run after parsing ---
|
|
33
|
+
def model_post_init(self, __context) -> None:
|
|
34
|
+
try:
|
|
35
|
+
s = (self.input or "").strip()
|
|
36
|
+
if not s:
|
|
37
|
+
raise ValueError("Input cannot be empty or whitespace")
|
|
38
|
+
|
|
39
|
+
# 1) Data URIs: validate & allow
|
|
40
|
+
if s.startswith("data:"):
|
|
41
|
+
mime = s.split(";", 1)[0].split(":", 1)[-1].lower()
|
|
42
|
+
if not (mime.startswith("audio/") or mime.startswith("image/")):
|
|
43
|
+
# Treat non-media data URIs as text – Protect supports only text/image/audio.
|
|
44
|
+
raise ValueError("Unsupported data URI mime; only audio/* or image/* allowed")
|
|
45
|
+
# Pass as-is
|
|
46
|
+
return
|
|
47
|
+
|
|
48
|
+
p = urlparse(s)
|
|
49
|
+
|
|
50
|
+
# 2) HTTP(S) URL: allow pass-through, but block known viewer pages
|
|
51
|
+
if p.scheme in ("http", "https"):
|
|
52
|
+
if self._is_blocked_url(s, p):
|
|
53
|
+
raise ValueError(
|
|
54
|
+
"This link looks like a preview page, not a direct file. "
|
|
55
|
+
"Use a direct download URL (e.g., raw.githubusercontent.com for GitHub; "
|
|
56
|
+
"export=download for Google Drive; dl.dropboxusercontent.com for Dropbox)."
|
|
57
|
+
)
|
|
58
|
+
# Do not sniff/transform: backend will handle it
|
|
59
|
+
return
|
|
60
|
+
|
|
61
|
+
# 3) Local file path → convert to data URI (fast, no network)
|
|
62
|
+
if self._looks_like_local_path(s):
|
|
63
|
+
ext = os.path.splitext(s)[1].lower()
|
|
64
|
+
if ext in self.AUDIO_EXTENSIONS:
|
|
65
|
+
self.input = self._file_to_data_uri(s, self._audio_mime(ext))
|
|
66
|
+
return
|
|
67
|
+
if ext in self.IMAGE_EXTENSIONS:
|
|
68
|
+
self.input = self._file_to_data_uri(s, self._image_mime(ext))
|
|
69
|
+
return
|
|
70
|
+
# Not a supported media extension
|
|
71
|
+
allowed_audio = ", ".join(sorted(self.AUDIO_EXTENSIONS))
|
|
72
|
+
allowed_img = ", ".join(sorted(self.IMAGE_EXTENSIONS))
|
|
73
|
+
raise ValueError(
|
|
74
|
+
f"Unsupported local file type '{ext}'. Supported audio: {allowed_audio}. "
|
|
75
|
+
f"Supported image: {allowed_img}."
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# 4) Otherwise: treat as plain text (backend handles as text)
|
|
79
|
+
# Nothing to change.
|
|
80
|
+
except Exception as e:
|
|
81
|
+
# Fail fast with a clean message. Protect supports only text/image/audio.
|
|
82
|
+
raise ValueError(f"Invalid Protect input: {e}") from e
|
|
83
|
+
|
|
84
|
+
@staticmethod
|
|
85
|
+
def _looks_like_local_path(s: str) -> bool:
|
|
86
|
+
# Absolute or relative file path that exists
|
|
87
|
+
try:
|
|
88
|
+
return os.path.exists(s)
|
|
89
|
+
except Exception:
|
|
90
|
+
return False
|
|
91
|
+
|
|
92
|
+
@staticmethod
|
|
93
|
+
def _is_blocked_url(s: str, p) -> bool:
|
|
94
|
+
host = (p.netloc or "").lower()
|
|
95
|
+
path = (p.path or "").lower()
|
|
96
|
+
|
|
97
|
+
# Known viewer / HTML pages that won't return raw bytes:
|
|
98
|
+
# GitHub blob pages
|
|
99
|
+
if host == "github.com" and "/blob/" in path:
|
|
100
|
+
return True
|
|
101
|
+
# Google Drive viewers
|
|
102
|
+
if host in ("drive.google.com", "docs.google.com") and ("/file/" in path or "/uc" in path) and "export=download" not in s:
|
|
103
|
+
return True
|
|
104
|
+
# Dropbox share pages (use dl.dropboxusercontent.com for direct)
|
|
105
|
+
if host == "www.dropbox.com" and "/s/" in path:
|
|
106
|
+
return True
|
|
107
|
+
# OneDrive viewer links
|
|
108
|
+
if host == "onedrive.live.com" and "redir" in path:
|
|
109
|
+
return True
|
|
110
|
+
|
|
111
|
+
return False
|
|
112
|
+
|
|
113
|
+
@staticmethod
|
|
114
|
+
def _audio_mime(ext: str) -> str:
|
|
115
|
+
return {
|
|
116
|
+
".mp3": "audio/mp3",
|
|
117
|
+
".wav": "audio/wav",
|
|
118
|
+
}.get(ext, "audio/mp3")
|
|
119
|
+
|
|
120
|
+
@staticmethod
|
|
121
|
+
def _image_mime(ext: str) -> str:
|
|
122
|
+
return {
|
|
123
|
+
".jpg": "image/jpeg",
|
|
124
|
+
".jpeg": "image/jpeg",
|
|
125
|
+
".png": "image/png",
|
|
126
|
+
".gif": "image/gif",
|
|
127
|
+
".webp": "image/webp",
|
|
128
|
+
".bmp": "image/bmp",
|
|
129
|
+
".tiff": "image/tiff",
|
|
130
|
+
".tif": "image/tiff",
|
|
131
|
+
".svg": "image/svg+xml",
|
|
132
|
+
}.get(ext, "image/jpeg")
|
|
133
|
+
|
|
134
|
+
def _file_to_data_uri(self, path: str, mime: str) -> str:
|
|
135
|
+
# Small, local-only work; no network. Guard against giant files.
|
|
136
|
+
size = os.path.getsize(path)
|
|
137
|
+
if size > self.MAX_LOCAL_BYTES:
|
|
138
|
+
raise ValueError(
|
|
139
|
+
f"Local file too large ({size} bytes). Max allowed is {self.MAX_LOCAL_BYTES}."
|
|
140
|
+
)
|
|
141
|
+
try:
|
|
142
|
+
with open(path, "rb") as f:
|
|
143
|
+
b64 = base64.b64encode(f.read()).decode("utf-8")
|
|
144
|
+
return f"data:{mime};base64,{b64}"
|
|
145
|
+
except Exception as e:
|
|
146
|
+
raise ValueError(f"Failed to read local file: {e}")
|
|
147
|
+
|
|
148
|
+
# An optional lightweight validator so callers get a nice error earlier if input is not a string.
|
|
149
|
+
@field_validator("input")
|
|
150
|
+
@classmethod
|
|
151
|
+
def _validate_input_is_str(cls, v: str) -> str:
|
|
152
|
+
if not isinstance(v, str):
|
|
153
|
+
raise TypeError("Protect input must be a string")
|
|
154
|
+
return v
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Streaming Evaluation Module.
|
|
2
|
+
|
|
3
|
+
Provides real-time evaluation of LLM outputs as tokens stream in,
|
|
4
|
+
with support for early stopping based on configurable policies.
|
|
5
|
+
|
|
6
|
+
Example:
|
|
7
|
+
from fi.evals.streaming import StreamingEvaluator, StreamingConfig, EarlyStopPolicy
|
|
8
|
+
|
|
9
|
+
# Create evaluator
|
|
10
|
+
evaluator = StreamingEvaluator(
|
|
11
|
+
config=StreamingConfig(
|
|
12
|
+
min_chunk_size=10,
|
|
13
|
+
max_chunk_size=100,
|
|
14
|
+
enable_early_stop=True,
|
|
15
|
+
),
|
|
16
|
+
policy=EarlyStopPolicy.default(),
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
# Add evaluation functions
|
|
20
|
+
evaluator.add_eval("toxicity", toxicity_scorer, threshold=0.7, pass_above=False)
|
|
21
|
+
|
|
22
|
+
# Process stream
|
|
23
|
+
for token in llm_stream:
|
|
24
|
+
result = evaluator.process_token(token)
|
|
25
|
+
if result and result.should_stop:
|
|
26
|
+
print(f"Early stop: {result.stop_reason}")
|
|
27
|
+
break
|
|
28
|
+
|
|
29
|
+
# Get final results
|
|
30
|
+
final_result = evaluator.finalize()
|
|
31
|
+
print(final_result.summary())
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from .types import (
|
|
35
|
+
ChunkResult,
|
|
36
|
+
EarlyStopCondition,
|
|
37
|
+
EarlyStopReason,
|
|
38
|
+
StreamingConfig,
|
|
39
|
+
StreamingEvalResult,
|
|
40
|
+
StreamingState,
|
|
41
|
+
)
|
|
42
|
+
from .buffer import BufferState, ChunkBuffer
|
|
43
|
+
from .policy import EarlyStopPolicy, PolicyState
|
|
44
|
+
from .evaluator import EvalSpec, StreamingEvaluator
|
|
45
|
+
from .scorers import (
|
|
46
|
+
toxicity_scorer,
|
|
47
|
+
safety_scorer,
|
|
48
|
+
pii_scorer,
|
|
49
|
+
jailbreak_scorer,
|
|
50
|
+
coherence_scorer,
|
|
51
|
+
quality_scorer,
|
|
52
|
+
safety_composite_scorer,
|
|
53
|
+
quality_composite_scorer,
|
|
54
|
+
create_keyword_scorer,
|
|
55
|
+
create_pattern_scorer,
|
|
56
|
+
CompositeScorer,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
__all__ = [
|
|
60
|
+
# Types
|
|
61
|
+
"ChunkResult",
|
|
62
|
+
"EarlyStopCondition",
|
|
63
|
+
"EarlyStopReason",
|
|
64
|
+
"StreamingConfig",
|
|
65
|
+
"StreamingEvalResult",
|
|
66
|
+
"StreamingState",
|
|
67
|
+
# Buffer
|
|
68
|
+
"BufferState",
|
|
69
|
+
"ChunkBuffer",
|
|
70
|
+
# Policy
|
|
71
|
+
"EarlyStopPolicy",
|
|
72
|
+
"PolicyState",
|
|
73
|
+
# Evaluator
|
|
74
|
+
"EvalSpec",
|
|
75
|
+
"StreamingEvaluator",
|
|
76
|
+
# Scorers
|
|
77
|
+
"toxicity_scorer",
|
|
78
|
+
"safety_scorer",
|
|
79
|
+
"pii_scorer",
|
|
80
|
+
"jailbreak_scorer",
|
|
81
|
+
"coherence_scorer",
|
|
82
|
+
"quality_scorer",
|
|
83
|
+
"safety_composite_scorer",
|
|
84
|
+
"quality_composite_scorer",
|
|
85
|
+
"create_keyword_scorer",
|
|
86
|
+
"create_pattern_scorer",
|
|
87
|
+
"CompositeScorer",
|
|
88
|
+
]
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Chunk buffer for streaming evaluation.
|
|
2
|
+
|
|
3
|
+
Accumulates tokens/chunks and manages when to trigger evaluations.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
import time
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Optional, Tuple
|
|
10
|
+
from .types import StreamingConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class BufferState:
|
|
15
|
+
"""Current state of the chunk buffer."""
|
|
16
|
+
|
|
17
|
+
total_text: str = ""
|
|
18
|
+
pending_text: str = ""
|
|
19
|
+
chunk_count: int = 0
|
|
20
|
+
token_count: int = 0
|
|
21
|
+
last_eval_time: float = 0.0
|
|
22
|
+
last_eval_position: int = 0
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ChunkBuffer:
|
|
26
|
+
"""
|
|
27
|
+
Buffer that accumulates streaming tokens and determines when to evaluate.
|
|
28
|
+
|
|
29
|
+
Manages the accumulation of tokens from a streaming LLM response and
|
|
30
|
+
decides when enough content has been collected to trigger an evaluation.
|
|
31
|
+
|
|
32
|
+
Example:
|
|
33
|
+
buffer = ChunkBuffer(config)
|
|
34
|
+
|
|
35
|
+
for token in stream:
|
|
36
|
+
buffer.add(token)
|
|
37
|
+
if buffer.should_evaluate():
|
|
38
|
+
chunk = buffer.get_chunk()
|
|
39
|
+
result = evaluator.evaluate_chunk(chunk, buffer.get_cumulative())
|
|
40
|
+
buffer.mark_evaluated()
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
# Sentence ending patterns
|
|
44
|
+
SENTENCE_ENDINGS = re.compile(r'[.!?]\s*$')
|
|
45
|
+
PARTIAL_SENTENCE = re.compile(r'[.!?]\s+')
|
|
46
|
+
|
|
47
|
+
def __init__(self, config: Optional[StreamingConfig] = None):
|
|
48
|
+
"""
|
|
49
|
+
Initialize the buffer.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
config: Streaming configuration (uses defaults if None)
|
|
53
|
+
"""
|
|
54
|
+
self.config = config or StreamingConfig()
|
|
55
|
+
self._state = BufferState()
|
|
56
|
+
self._start_time = time.perf_counter()
|
|
57
|
+
|
|
58
|
+
def add(self, token: str) -> None:
|
|
59
|
+
"""
|
|
60
|
+
Add a token to the buffer.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
token: The token text to add
|
|
64
|
+
"""
|
|
65
|
+
self._state.total_text += token
|
|
66
|
+
self._state.pending_text += token
|
|
67
|
+
self._state.token_count += 1
|
|
68
|
+
|
|
69
|
+
def add_chunk(self, chunk: str) -> None:
|
|
70
|
+
"""
|
|
71
|
+
Add a larger chunk of text to the buffer.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
chunk: The chunk text to add
|
|
75
|
+
"""
|
|
76
|
+
self._state.total_text += chunk
|
|
77
|
+
self._state.pending_text += chunk
|
|
78
|
+
# Estimate token count (rough approximation)
|
|
79
|
+
self._state.token_count += len(chunk.split())
|
|
80
|
+
|
|
81
|
+
def should_evaluate(self) -> bool:
|
|
82
|
+
"""
|
|
83
|
+
Check if we should trigger an evaluation.
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
True if evaluation should be triggered
|
|
87
|
+
"""
|
|
88
|
+
pending_len = len(self._state.pending_text)
|
|
89
|
+
|
|
90
|
+
# Check minimum chunk size
|
|
91
|
+
if pending_len < self.config.min_chunk_size:
|
|
92
|
+
return False
|
|
93
|
+
|
|
94
|
+
# Check time interval
|
|
95
|
+
current_time = time.perf_counter()
|
|
96
|
+
time_since_last = (current_time - self._state.last_eval_time) * 1000
|
|
97
|
+
|
|
98
|
+
# Force evaluation if max chunk size reached
|
|
99
|
+
if pending_len >= self.config.max_chunk_size:
|
|
100
|
+
return True
|
|
101
|
+
|
|
102
|
+
# Check time-based interval
|
|
103
|
+
if time_since_last < self.config.eval_interval_ms:
|
|
104
|
+
return False
|
|
105
|
+
|
|
106
|
+
# Check sentence boundary if enabled
|
|
107
|
+
if self.config.eval_on_sentence_end:
|
|
108
|
+
if self.SENTENCE_ENDINGS.search(self._state.pending_text):
|
|
109
|
+
return True
|
|
110
|
+
|
|
111
|
+
# Check chunk count interval
|
|
112
|
+
if self.config.eval_every_n_chunks > 1:
|
|
113
|
+
# Only evaluate every N chunks
|
|
114
|
+
next_chunk = self._state.chunk_count + 1
|
|
115
|
+
if next_chunk % self.config.eval_every_n_chunks != 0:
|
|
116
|
+
return pending_len >= self.config.max_chunk_size
|
|
117
|
+
|
|
118
|
+
return pending_len >= self.config.min_chunk_size
|
|
119
|
+
|
|
120
|
+
def get_chunk(self) -> str:
|
|
121
|
+
"""
|
|
122
|
+
Get the pending chunk for evaluation.
|
|
123
|
+
|
|
124
|
+
Returns:
|
|
125
|
+
The pending text that should be evaluated
|
|
126
|
+
"""
|
|
127
|
+
return self._state.pending_text
|
|
128
|
+
|
|
129
|
+
def get_cumulative(self) -> str:
|
|
130
|
+
"""
|
|
131
|
+
Get all accumulated text so far.
|
|
132
|
+
|
|
133
|
+
Returns:
|
|
134
|
+
The complete accumulated text
|
|
135
|
+
"""
|
|
136
|
+
return self._state.total_text
|
|
137
|
+
|
|
138
|
+
def mark_evaluated(self) -> None:
|
|
139
|
+
"""Mark the current pending content as evaluated."""
|
|
140
|
+
self._state.chunk_count += 1
|
|
141
|
+
self._state.last_eval_time = time.perf_counter()
|
|
142
|
+
self._state.last_eval_position = len(self._state.total_text)
|
|
143
|
+
self._state.pending_text = ""
|
|
144
|
+
|
|
145
|
+
def get_chunk_index(self) -> int:
|
|
146
|
+
"""Get the current chunk index."""
|
|
147
|
+
return self._state.chunk_count
|
|
148
|
+
|
|
149
|
+
def get_token_count(self) -> int:
|
|
150
|
+
"""Get the total token count."""
|
|
151
|
+
return self._state.token_count
|
|
152
|
+
|
|
153
|
+
def get_char_count(self) -> int:
|
|
154
|
+
"""Get the total character count."""
|
|
155
|
+
return len(self._state.total_text)
|
|
156
|
+
|
|
157
|
+
def should_stop_for_limits(self) -> Tuple[bool, str]:
|
|
158
|
+
"""
|
|
159
|
+
Check if we should stop due to limits.
|
|
160
|
+
|
|
161
|
+
Returns:
|
|
162
|
+
Tuple of (should_stop, reason)
|
|
163
|
+
"""
|
|
164
|
+
# Check token limit
|
|
165
|
+
if self.config.max_tokens and self._state.token_count >= self.config.max_tokens:
|
|
166
|
+
return True, "max_tokens"
|
|
167
|
+
|
|
168
|
+
# Check character limit
|
|
169
|
+
if self.config.max_chars and len(self._state.total_text) >= self.config.max_chars:
|
|
170
|
+
return True, "max_chars"
|
|
171
|
+
|
|
172
|
+
# Check total timeout
|
|
173
|
+
elapsed_ms = (time.perf_counter() - self._start_time) * 1000
|
|
174
|
+
if elapsed_ms >= self.config.total_timeout_ms:
|
|
175
|
+
return True, "timeout"
|
|
176
|
+
|
|
177
|
+
return False, ""
|
|
178
|
+
|
|
179
|
+
def reset(self) -> None:
|
|
180
|
+
"""Reset the buffer to initial state."""
|
|
181
|
+
self._state = BufferState()
|
|
182
|
+
self._start_time = time.perf_counter()
|
|
183
|
+
|
|
184
|
+
@property
|
|
185
|
+
def state(self) -> BufferState:
|
|
186
|
+
"""Get the current buffer state."""
|
|
187
|
+
return self._state
|
|
188
|
+
|
|
189
|
+
@property
|
|
190
|
+
def is_empty(self) -> bool:
|
|
191
|
+
"""Check if buffer has no content."""
|
|
192
|
+
return len(self._state.total_text) == 0
|
|
193
|
+
|
|
194
|
+
@property
|
|
195
|
+
def has_pending(self) -> bool:
|
|
196
|
+
"""Check if there's pending unevaluated content."""
|
|
197
|
+
return len(self._state.pending_text) > 0
|
|
198
|
+
|
|
199
|
+
def get_stats(self) -> dict:
|
|
200
|
+
"""Get buffer statistics."""
|
|
201
|
+
elapsed_ms = (time.perf_counter() - self._start_time) * 1000
|
|
202
|
+
return {
|
|
203
|
+
"total_chars": len(self._state.total_text),
|
|
204
|
+
"pending_chars": len(self._state.pending_text),
|
|
205
|
+
"chunk_count": self._state.chunk_count,
|
|
206
|
+
"token_count": self._state.token_count,
|
|
207
|
+
"elapsed_ms": elapsed_ms,
|
|
208
|
+
"avg_chunk_size": (
|
|
209
|
+
len(self._state.total_text) / self._state.chunk_count
|
|
210
|
+
if self._state.chunk_count > 0
|
|
211
|
+
else 0
|
|
212
|
+
),
|
|
213
|
+
}
|