agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
import optuna
|
|
2
|
+
import logging
|
|
3
|
+
import random
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
import time
|
|
7
|
+
from typing import List, Dict, Any, Optional, Callable
|
|
8
|
+
from ..base.base_optimizer import BaseOptimizer
|
|
9
|
+
from ..types import OptimizationResult, IterationHistory, EvaluationResult
|
|
10
|
+
from ..datamappers import BasicDataMapper
|
|
11
|
+
from ..generators.litellm import LiteLLMGenerator
|
|
12
|
+
from ..base.evaluator import Evaluator
|
|
13
|
+
from ..utils.early_stopping import EarlyStoppingConfig, EarlyStoppingChecker
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
TEACHER_SYSTEM_PROMPT = (
|
|
17
|
+
"""
|
|
18
|
+
You are an expert prompt engineer with deep knowledge of few-shot learning and template design. Your task is to analyze a sample of dataset items and create an optimal Python .format() string template for few-shot examples.
|
|
19
|
+
|
|
20
|
+
ANALYSIS REQUIREMENTS:
|
|
21
|
+
1. Examine the structure and content of the provided dataset examples
|
|
22
|
+
2. Identify all available field names/keys in the examples
|
|
23
|
+
3. Determine which fields represent inputs vs. expected outputs
|
|
24
|
+
4. Design a template that clearly demonstrates the input-output relationship
|
|
25
|
+
|
|
26
|
+
TEMPLATE DESIGN PRINCIPLES:
|
|
27
|
+
- Use ONLY field names that actually exist in the provided examples
|
|
28
|
+
- Include both input and output fields to enable effective few-shot learning
|
|
29
|
+
- Create clear, readable formatting that helps models understand the pattern
|
|
30
|
+
- Use descriptive labels (e.g., "Input:", "Output:", "Question:", "Answer:")
|
|
31
|
+
- Ensure the template is concise yet informative
|
|
32
|
+
- Maintain consistent formatting across examples
|
|
33
|
+
|
|
34
|
+
OUTPUT FORMAT:
|
|
35
|
+
Return ONLY a valid JSON object with this exact structure:
|
|
36
|
+
{
|
|
37
|
+
"example_template": "your_template_string_here"
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
The template string must:
|
|
41
|
+
- Use Python .format() syntax with curly braces for field substitution
|
|
42
|
+
- Include clear labels for input and output sections
|
|
43
|
+
- Be ready to use without any modifications
|
|
44
|
+
- Work for all examples in the dataset
|
|
45
|
+
|
|
46
|
+
Example of a well-formed template:
|
|
47
|
+
"Question: {question}\nAnswer: {answer}"
|
|
48
|
+
or
|
|
49
|
+
"Prompt: {prompt}\nExpected Response: {response}\n---"
|
|
50
|
+
|
|
51
|
+
DO NOT include any explanations, comments, or additional text - only the JSON object.
|
|
52
|
+
"""
|
|
53
|
+
).strip()
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class BayesianSearchOptimizer(BaseOptimizer):
|
|
57
|
+
"""
|
|
58
|
+
An optimizer that uses Bayesian optimization (via Optuna) to find the
|
|
59
|
+
best prompt by intelligently selecting few-shot examples.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
def __init__(
|
|
63
|
+
self,
|
|
64
|
+
# Few-shot search space
|
|
65
|
+
min_examples: int = 2,
|
|
66
|
+
max_examples: int = 8,
|
|
67
|
+
allow_repeats: bool = False,
|
|
68
|
+
fixed_example_indices: Optional[List[int]] = None,
|
|
69
|
+
# Trials and randomness
|
|
70
|
+
n_trials: int = 10,
|
|
71
|
+
seed: int = 42,
|
|
72
|
+
# Inference/generation config
|
|
73
|
+
inference_model_name: str = "gpt-4o-mini",
|
|
74
|
+
inference_model_kwargs: Optional[Dict[str, Any]] = None,
|
|
75
|
+
# Example formatting and prompt construction
|
|
76
|
+
example_template: Optional[str] = None,
|
|
77
|
+
example_template_fields: Optional[List[str]] = None,
|
|
78
|
+
field_aliases: Optional[Dict[str, str]] = None,
|
|
79
|
+
example_separator: str = "\n",
|
|
80
|
+
few_shot_position: str = "append", # "prepend" | "append"
|
|
81
|
+
prompt_builder: Optional[Callable[[str, List[str]], str]] = None,
|
|
82
|
+
example_formatter: Optional[Callable[[Dict[str, Any]], str]] = None,
|
|
83
|
+
few_shot_title: Optional[str] = None,
|
|
84
|
+
# Teacher-guided template inference (optional)
|
|
85
|
+
infer_example_template_via_teacher: bool = False,
|
|
86
|
+
teacher_model_name: str = "gpt-5",
|
|
87
|
+
teacher_model_kwargs: Optional[Dict[str, Any]] = None,
|
|
88
|
+
template_infer_n_samples: int = 8,
|
|
89
|
+
teacher_system_prompt: str = TEACHER_SYSTEM_PROMPT,
|
|
90
|
+
teacher_infer_max_retries: int = 2,
|
|
91
|
+
teacher_infer_retry_sleep: float = 0.5,
|
|
92
|
+
# Evaluation controls
|
|
93
|
+
eval_subset_size: Optional[int] = None,
|
|
94
|
+
eval_subset_strategy: str = "random", # "random" | "first" | "all"
|
|
95
|
+
score_aggregator: Optional[Callable[[List[EvaluationResult]], float]] = None,
|
|
96
|
+
# Optuna controls
|
|
97
|
+
sampler: Optional[optuna.samplers.BaseSampler] = None,
|
|
98
|
+
pruner: Optional[optuna.pruners.BasePruner] = None,
|
|
99
|
+
direction: str = "maximize",
|
|
100
|
+
storage: Optional[str] = None,
|
|
101
|
+
study_name: Optional[str] = None,
|
|
102
|
+
):
|
|
103
|
+
# Search space
|
|
104
|
+
self.min_examples = min_examples
|
|
105
|
+
self.max_examples = max_examples
|
|
106
|
+
self.allow_repeats = allow_repeats
|
|
107
|
+
self.fixed_example_indices = fixed_example_indices or []
|
|
108
|
+
# Trials and randomness
|
|
109
|
+
self.n_trials = n_trials
|
|
110
|
+
self.seed = seed
|
|
111
|
+
# Inference/generation
|
|
112
|
+
self.inference_model_name = inference_model_name
|
|
113
|
+
self.inference_model_kwargs = inference_model_kwargs or {}
|
|
114
|
+
# Formatting/building
|
|
115
|
+
self.example_template = example_template
|
|
116
|
+
self.example_template_fields = example_template_fields
|
|
117
|
+
self.field_aliases = field_aliases or {}
|
|
118
|
+
self.example_separator = example_separator
|
|
119
|
+
self.few_shot_position = few_shot_position
|
|
120
|
+
self.prompt_builder = prompt_builder
|
|
121
|
+
self.example_formatter = example_formatter
|
|
122
|
+
self.few_shot_title = few_shot_title
|
|
123
|
+
# Teacher-guided template inference
|
|
124
|
+
self.infer_example_template_via_teacher = infer_example_template_via_teacher
|
|
125
|
+
self.teacher_model_name = teacher_model_name
|
|
126
|
+
# default kwargs for gpt-5 style models
|
|
127
|
+
default_teacher_kwargs: Dict[str, Any] = {
|
|
128
|
+
"temperature": 1.0,
|
|
129
|
+
"max_tokens": 16000,
|
|
130
|
+
}
|
|
131
|
+
self.teacher_model_kwargs = {
|
|
132
|
+
**default_teacher_kwargs,
|
|
133
|
+
**(teacher_model_kwargs or {}),
|
|
134
|
+
}
|
|
135
|
+
self.template_infer_n_samples = template_infer_n_samples
|
|
136
|
+
self.teacher_system_prompt = teacher_system_prompt
|
|
137
|
+
self.teacher_infer_max_retries = max(0, int(teacher_infer_max_retries))
|
|
138
|
+
self.teacher_infer_retry_sleep = max(0.0, float(teacher_infer_retry_sleep))
|
|
139
|
+
# Evaluation
|
|
140
|
+
self.eval_subset_size = eval_subset_size
|
|
141
|
+
self.eval_subset_strategy = eval_subset_strategy
|
|
142
|
+
self.score_aggregator = score_aggregator or self._default_score_aggregator
|
|
143
|
+
# Optuna
|
|
144
|
+
self.sampler = sampler or optuna.samplers.TPESampler(seed=self.seed)
|
|
145
|
+
self.pruner = pruner
|
|
146
|
+
self.direction = direction
|
|
147
|
+
self.storage = storage
|
|
148
|
+
self.study_name = study_name
|
|
149
|
+
# runtime state
|
|
150
|
+
self._runtime_example_template: Optional[str] = None
|
|
151
|
+
|
|
152
|
+
def optimize(
|
|
153
|
+
self,
|
|
154
|
+
evaluator: Evaluator,
|
|
155
|
+
data_mapper: BasicDataMapper,
|
|
156
|
+
dataset: List[Dict[str, Any]],
|
|
157
|
+
initial_prompts: List[str],
|
|
158
|
+
early_stopping: Optional[EarlyStoppingConfig] = None,
|
|
159
|
+
**kwargs: Any,
|
|
160
|
+
) -> OptimizationResult:
|
|
161
|
+
logging.info("--- Starting Bayesian Search Optimization ---")
|
|
162
|
+
|
|
163
|
+
# Initialize early stopping checker
|
|
164
|
+
checker = None
|
|
165
|
+
if early_stopping and early_stopping.is_enabled():
|
|
166
|
+
checker = EarlyStoppingChecker(early_stopping)
|
|
167
|
+
logging.info(f"Early stopping enabled: {early_stopping}")
|
|
168
|
+
|
|
169
|
+
if not initial_prompts:
|
|
170
|
+
raise ValueError("Initial prompts list cannot be empty.")
|
|
171
|
+
|
|
172
|
+
initial_prompt = initial_prompts[0]
|
|
173
|
+
history: List[IterationHistory] = []
|
|
174
|
+
|
|
175
|
+
# Optionally infer the example template via a teacher model from a sample of the dataset
|
|
176
|
+
self._runtime_example_template = None
|
|
177
|
+
if self.infer_example_template_via_teacher:
|
|
178
|
+
try:
|
|
179
|
+
self._runtime_example_template = self._infer_example_template(dataset)
|
|
180
|
+
logging.info(
|
|
181
|
+
f"Inferred example template via teacher model: \n {self._runtime_example_template}"
|
|
182
|
+
)
|
|
183
|
+
except Exception as e:
|
|
184
|
+
logging.warning(f"Falling back to default example_template. Error: {e}")
|
|
185
|
+
self._runtime_example_template = None
|
|
186
|
+
|
|
187
|
+
def objective(trial: optuna.Trial) -> float:
|
|
188
|
+
# Suggest number of few-shot examples
|
|
189
|
+
n_examples = trial.suggest_int(
|
|
190
|
+
"n_examples", self.min_examples, self.max_examples
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
# Use a single seed to derive indices, avoiding dynamic value spaces in Optuna
|
|
194
|
+
example_seed = trial.suggest_int("example_seed", 0, 2_000_000_000)
|
|
195
|
+
rng = random.Random(example_seed)
|
|
196
|
+
|
|
197
|
+
# Honor fixed indices first
|
|
198
|
+
selected_indices: List[int] = list(self.fixed_example_indices)
|
|
199
|
+
remaining_needed = max(0, n_examples - len(selected_indices))
|
|
200
|
+
|
|
201
|
+
if remaining_needed > 0:
|
|
202
|
+
if self.allow_repeats:
|
|
203
|
+
# Repeats allowed: sample with replacement
|
|
204
|
+
more = [
|
|
205
|
+
rng.randrange(len(dataset)) for _ in range(remaining_needed)
|
|
206
|
+
]
|
|
207
|
+
selected_indices.extend(more)
|
|
208
|
+
else:
|
|
209
|
+
# Unique sampling: sample without replacement from remaining pool
|
|
210
|
+
pool = [
|
|
211
|
+
i for i in range(len(dataset)) if i not in set(selected_indices)
|
|
212
|
+
]
|
|
213
|
+
take = min(remaining_needed, len(pool))
|
|
214
|
+
selected_indices.extend(rng.sample(pool, take))
|
|
215
|
+
|
|
216
|
+
# Format the selected examples for few-shot
|
|
217
|
+
demo_examples = [dataset[i] for i in selected_indices]
|
|
218
|
+
example_strings = [self._format_example(ex) for ex in demo_examples]
|
|
219
|
+
few_shot_block = self._build_few_shot_block(example_strings)
|
|
220
|
+
|
|
221
|
+
# Build the full prompt
|
|
222
|
+
full_prompt = self._build_prompt(initial_prompt, few_shot_block)
|
|
223
|
+
|
|
224
|
+
# Score the prompt
|
|
225
|
+
iteration_history = self._score_prompt(
|
|
226
|
+
full_prompt, evaluator, data_mapper, dataset
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
if not iteration_history:
|
|
230
|
+
trial.set_user_attr("prompt", full_prompt)
|
|
231
|
+
return 0.0
|
|
232
|
+
|
|
233
|
+
history.append(iteration_history)
|
|
234
|
+
avg_score = iteration_history.average_score
|
|
235
|
+
trial.set_user_attr("prompt", full_prompt)
|
|
236
|
+
logging.info(
|
|
237
|
+
f"Trial {trial.number}: Score={avg_score:.4f}, Num Examples={len(selected_indices)}"
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
# Check early stopping
|
|
241
|
+
if checker:
|
|
242
|
+
eval_size = len(self._select_eval_subset(dataset))
|
|
243
|
+
if checker.should_stop(avg_score, eval_size):
|
|
244
|
+
logging.info(
|
|
245
|
+
f"Early stopping triggered: {checker.get_state()['stop_reason']}"
|
|
246
|
+
)
|
|
247
|
+
trial.study.stop()
|
|
248
|
+
|
|
249
|
+
return avg_score
|
|
250
|
+
|
|
251
|
+
study = optuna.create_study(
|
|
252
|
+
direction=self.direction,
|
|
253
|
+
sampler=self.sampler,
|
|
254
|
+
pruner=self.pruner,
|
|
255
|
+
storage=self.storage,
|
|
256
|
+
study_name=self.study_name,
|
|
257
|
+
load_if_exists=bool(self.storage and self.study_name),
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
try:
|
|
261
|
+
study.optimize(objective, n_trials=self.n_trials)
|
|
262
|
+
except Exception as e:
|
|
263
|
+
logging.info(f"Optimization stopped: {e}")
|
|
264
|
+
|
|
265
|
+
# Check if any trials completed before accessing best_trial
|
|
266
|
+
if not history:
|
|
267
|
+
raise RuntimeError(
|
|
268
|
+
"Optimization stopped before any trials completed successfully"
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
best_prompt = study.best_trial.user_attrs.get("prompt", initial_prompt)
|
|
272
|
+
best_generator = LiteLLMGenerator(self.inference_model_name, best_prompt)
|
|
273
|
+
|
|
274
|
+
# Build result with early stopping metadata
|
|
275
|
+
return OptimizationResult(
|
|
276
|
+
best_generator=best_generator,
|
|
277
|
+
history=history,
|
|
278
|
+
final_score=float(study.best_value)
|
|
279
|
+
if study.best_value is not None
|
|
280
|
+
else 0.0,
|
|
281
|
+
early_stopped=checker.get_state()["stopped"] if checker else False,
|
|
282
|
+
stop_reason=checker.get_state()["stop_reason"] if checker else None,
|
|
283
|
+
total_iterations=len(history),
|
|
284
|
+
total_evaluations=(
|
|
285
|
+
checker.get_state()["total_evaluations"]
|
|
286
|
+
if checker
|
|
287
|
+
else sum(len(h.individual_results) for h in history)
|
|
288
|
+
),
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
def _score_prompt(
|
|
292
|
+
self,
|
|
293
|
+
prompt: str,
|
|
294
|
+
evaluator: Evaluator,
|
|
295
|
+
data_mapper: BasicDataMapper,
|
|
296
|
+
dataset: List[Dict[str, Any]],
|
|
297
|
+
) -> Optional[IterationHistory]:
|
|
298
|
+
try:
|
|
299
|
+
eval_dataset = self._select_eval_subset(dataset)
|
|
300
|
+
temp_generator = LiteLLMGenerator(self.inference_model_name, prompt)
|
|
301
|
+
|
|
302
|
+
generated_outputs = [
|
|
303
|
+
temp_generator.generate(example, **self.inference_model_kwargs)
|
|
304
|
+
for example in eval_dataset
|
|
305
|
+
]
|
|
306
|
+
eval_inputs = [
|
|
307
|
+
data_mapper.map(gen_out, ex)
|
|
308
|
+
for gen_out, ex in zip(generated_outputs, eval_dataset)
|
|
309
|
+
]
|
|
310
|
+
results = evaluator.evaluate(eval_inputs)
|
|
311
|
+
avg_score = self.score_aggregator(results)
|
|
312
|
+
return IterationHistory(
|
|
313
|
+
prompt=prompt, average_score=avg_score, individual_results=results
|
|
314
|
+
)
|
|
315
|
+
except Exception as e:
|
|
316
|
+
logging.error(f"Failed to score prompt: {e}")
|
|
317
|
+
return None
|
|
318
|
+
|
|
319
|
+
def _infer_example_template(self, dataset: List[Dict[str, Any]]) -> str:
|
|
320
|
+
sample_size = min(self.template_infer_n_samples, max(1, len(dataset)))
|
|
321
|
+
sample = (
|
|
322
|
+
random.sample(dataset, sample_size)
|
|
323
|
+
if len(dataset) > sample_size
|
|
324
|
+
else dataset
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
# Build a minimal payload: include keys and a few short examples limited to those keys
|
|
328
|
+
keys: List[str] = sorted({k for ex in sample for k in ex.keys()})
|
|
329
|
+
trimmed_examples: List[Dict[str, Any]] = [
|
|
330
|
+
{k: str(ex.get(k, ""))[:500] for k in keys} for ex in sample
|
|
331
|
+
]
|
|
332
|
+
user_payload = json.dumps(
|
|
333
|
+
{"keys": keys, "examples": trimmed_examples}, ensure_ascii=False
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
prompt_template = (
|
|
337
|
+
f"{self.teacher_system_prompt}\n\n"
|
|
338
|
+
"Available keys:\n{keys}\n\n"
|
|
339
|
+
"Examples (JSON):\n{examples_json}\n\n"
|
|
340
|
+
'Respond ONLY with a JSON object like {{"example_template": "..."}}.'
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
teacher = LiteLLMGenerator(self.teacher_model_name, prompt_template)
|
|
344
|
+
|
|
345
|
+
last_err: Optional[Exception] = None
|
|
346
|
+
for attempt in range(self.teacher_infer_max_retries + 1):
|
|
347
|
+
try:
|
|
348
|
+
content = teacher.generate(
|
|
349
|
+
{"keys": ", ".join(keys), "examples_json": user_payload},
|
|
350
|
+
response_format={"type": "json_object"},
|
|
351
|
+
**self.teacher_model_kwargs,
|
|
352
|
+
)
|
|
353
|
+
template = self._parse_example_template_from_content(content)
|
|
354
|
+
if template:
|
|
355
|
+
return template
|
|
356
|
+
raise ValueError("Missing or empty 'example_template' in response")
|
|
357
|
+
except Exception as e:
|
|
358
|
+
last_err = e
|
|
359
|
+
if attempt < self.teacher_infer_max_retries:
|
|
360
|
+
time.sleep(self.teacher_infer_retry_sleep)
|
|
361
|
+
else:
|
|
362
|
+
break
|
|
363
|
+
raise RuntimeError(f"Teacher template inference failed: {last_err}")
|
|
364
|
+
|
|
365
|
+
@staticmethod
|
|
366
|
+
def _parse_example_template_from_content(content: str) -> Optional[str]:
|
|
367
|
+
# First try strict JSON
|
|
368
|
+
try:
|
|
369
|
+
data = json.loads(content)
|
|
370
|
+
tmpl = data.get("example_template")
|
|
371
|
+
if isinstance(tmpl, str) and tmpl.strip():
|
|
372
|
+
return tmpl
|
|
373
|
+
except Exception:
|
|
374
|
+
pass
|
|
375
|
+
# Try to extract JSON object containing example_template
|
|
376
|
+
try:
|
|
377
|
+
match = re.search(
|
|
378
|
+
r"\{[\s\S]*?\"example_template\"\s*:\s*\"[\s\S]*?\"[\s\S]*?\}", content
|
|
379
|
+
)
|
|
380
|
+
if match:
|
|
381
|
+
obj = json.loads(match.group(0))
|
|
382
|
+
tmpl = obj.get("example_template")
|
|
383
|
+
if isinstance(tmpl, str) and tmpl.strip():
|
|
384
|
+
return tmpl
|
|
385
|
+
except Exception:
|
|
386
|
+
pass
|
|
387
|
+
return None
|
|
388
|
+
|
|
389
|
+
def _select_eval_subset(
|
|
390
|
+
self, dataset: List[Dict[str, Any]]
|
|
391
|
+
) -> List[Dict[str, Any]]:
|
|
392
|
+
if not self.eval_subset_size or self.eval_subset_size >= len(dataset):
|
|
393
|
+
return dataset
|
|
394
|
+
size = max(1, self.eval_subset_size)
|
|
395
|
+
if self.eval_subset_strategy == "first":
|
|
396
|
+
return dataset[:size]
|
|
397
|
+
elif self.eval_subset_strategy == "random":
|
|
398
|
+
return random.sample(dataset, size)
|
|
399
|
+
else:
|
|
400
|
+
return dataset
|
|
401
|
+
|
|
402
|
+
def _format_example(self, example: Dict[str, Any]) -> str:
|
|
403
|
+
if self.example_formatter:
|
|
404
|
+
return self.example_formatter(example)
|
|
405
|
+
template = self._runtime_example_template or self.example_template
|
|
406
|
+
if template:
|
|
407
|
+
try:
|
|
408
|
+
return template.format(**example)
|
|
409
|
+
except Exception:
|
|
410
|
+
pass
|
|
411
|
+
# Fallbacks when no template or failed formatting
|
|
412
|
+
if self.example_template_fields:
|
|
413
|
+
lines: List[str] = []
|
|
414
|
+
for key in self.example_template_fields:
|
|
415
|
+
if key in example:
|
|
416
|
+
label = self.field_aliases.get(key, key)
|
|
417
|
+
lines.append(f"{label}: {example[key]}")
|
|
418
|
+
if lines:
|
|
419
|
+
return "\n".join(lines)
|
|
420
|
+
# Final fallback: JSON dump of the example
|
|
421
|
+
return json.dumps(example, ensure_ascii=False)
|
|
422
|
+
|
|
423
|
+
def _build_few_shot_block(self, example_strings: List[str]) -> str:
|
|
424
|
+
block = self.example_separator.join(example_strings)
|
|
425
|
+
if self.few_shot_title:
|
|
426
|
+
return f"{self.few_shot_title}\n{block}"
|
|
427
|
+
return block
|
|
428
|
+
|
|
429
|
+
def _build_prompt(self, base_prompt: str, few_shot_block: str) -> str:
|
|
430
|
+
if self.prompt_builder:
|
|
431
|
+
return self.prompt_builder(base_prompt, [few_shot_block])
|
|
432
|
+
if not few_shot_block:
|
|
433
|
+
return base_prompt
|
|
434
|
+
# Escape braces in few-shot block to avoid str.format collisions
|
|
435
|
+
safe_block = self._escape_braces(few_shot_block)
|
|
436
|
+
if self.few_shot_position == "prepend":
|
|
437
|
+
return f"{safe_block}\n\n---\n\n{base_prompt}"
|
|
438
|
+
# default append
|
|
439
|
+
return f"{base_prompt}\n\n---\n\n{safe_block}\n\n---"
|
|
440
|
+
|
|
441
|
+
@staticmethod
|
|
442
|
+
def _escape_braces(text: str) -> str:
|
|
443
|
+
return text.replace("{", "{{").replace("}", "}}")
|
|
444
|
+
|
|
445
|
+
@staticmethod
|
|
446
|
+
def _default_score_aggregator(results: List[EvaluationResult]) -> float:
|
|
447
|
+
if not results:
|
|
448
|
+
return 0.0
|
|
449
|
+
return sum(r.score for r in results) / max(1, len(results))
|