agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import time
|
|
3
|
+
from typing import Any, Dict, List, Optional
|
|
4
|
+
|
|
5
|
+
# Import GEPA's core components. A try/except block makes this a soft dependency.
|
|
6
|
+
try:
|
|
7
|
+
import gepa
|
|
8
|
+
from gepa.core.adapter import GEPAAdapter, EvaluationBatch, DataInst
|
|
9
|
+
except ImportError:
|
|
10
|
+
raise ImportError(
|
|
11
|
+
"To use GEPAOptimizer, please install the 'gepa' library with: pip install gepa"
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
from ..base.base_optimizer import BaseOptimizer
|
|
15
|
+
from ..datamappers.basic_mapper import BasicDataMapper
|
|
16
|
+
from ..base.evaluator import Evaluator
|
|
17
|
+
from ..generators.litellm import LiteLLMGenerator
|
|
18
|
+
from ..types import OptimizationResult, IterationHistory
|
|
19
|
+
from ..utils.early_stopping import (
|
|
20
|
+
EarlyStoppingConfig,
|
|
21
|
+
EarlyStoppingChecker,
|
|
22
|
+
EarlyStoppingException,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
logger = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class _InternalGEPAAdapter(GEPAAdapter[DataInst, Dict[str, Any], Dict[str, Any]]):
|
|
29
|
+
"""
|
|
30
|
+
An internal adapter that translates our framework's components (Evaluator,
|
|
31
|
+
DataMapper) into the interface GEPA's optimization engine expects.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(
|
|
35
|
+
self,
|
|
36
|
+
generator_model: str,
|
|
37
|
+
evaluator: Evaluator,
|
|
38
|
+
data_mapper: BasicDataMapper,
|
|
39
|
+
history_list: List[IterationHistory],
|
|
40
|
+
early_stopping_checker: Optional[EarlyStoppingChecker] = None,
|
|
41
|
+
):
|
|
42
|
+
self.generator_model = generator_model
|
|
43
|
+
self.evaluator = evaluator
|
|
44
|
+
self.data_mapper = data_mapper
|
|
45
|
+
self.history_list = history_list
|
|
46
|
+
self.early_stopping_checker = early_stopping_checker
|
|
47
|
+
logger.info(f"Initialized with generator_model: {generator_model}")
|
|
48
|
+
|
|
49
|
+
def evaluate(
|
|
50
|
+
self,
|
|
51
|
+
batch: List[Dict[str, Any]],
|
|
52
|
+
candidate: Dict[str, str],
|
|
53
|
+
capture_traces: bool = False,
|
|
54
|
+
) -> EvaluationBatch[Dict[str, Any], Dict[str, Any]]:
|
|
55
|
+
"""
|
|
56
|
+
This method is called by GEPA during its optimization loop. It uses
|
|
57
|
+
our framework's components to perform the evaluation.
|
|
58
|
+
"""
|
|
59
|
+
eval_start_time = time.time()
|
|
60
|
+
logger.info("Starting evaluation for a candidate prompt.")
|
|
61
|
+
|
|
62
|
+
# GEPA provides the prompt as the first (and only) value in the candidate dict.
|
|
63
|
+
prompt_text = next(iter(candidate.values()))
|
|
64
|
+
logger.info(f"Evaluating prompt: '{prompt_text[:100]}...'")
|
|
65
|
+
logger.info(f"Batch size: {len(batch)}")
|
|
66
|
+
|
|
67
|
+
temp_generator = LiteLLMGenerator(
|
|
68
|
+
model=self.generator_model, prompt_template=prompt_text
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
logger.info("Generating outputs...")
|
|
72
|
+
gen_start_time = time.time()
|
|
73
|
+
generated_outputs = [temp_generator.generate(example) for example in batch]
|
|
74
|
+
gen_end_time = time.time()
|
|
75
|
+
logger.info(
|
|
76
|
+
f"Output generation finished in {gen_end_time - gen_start_time:.2f}s."
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
logger.info("Mapping evaluation inputs...")
|
|
80
|
+
eval_inputs = [
|
|
81
|
+
self.data_mapper.map(gen_out, ex)
|
|
82
|
+
for gen_out, ex in zip(generated_outputs, batch)
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
logger.info("Evaluating generated outputs...")
|
|
86
|
+
evaluator_start_time = time.time()
|
|
87
|
+
results = self.evaluator.evaluate(eval_inputs)
|
|
88
|
+
evaluator_end_time = time.time()
|
|
89
|
+
logger.info(
|
|
90
|
+
f"Evaluation with framework evaluator finished in {evaluator_end_time - evaluator_start_time:.2f}s."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
scores = [res.score for res in results]
|
|
94
|
+
logger.info(f"Scores: {scores}")
|
|
95
|
+
outputs = [
|
|
96
|
+
{"generated_text": out, "full_result": res.model_dump()}
|
|
97
|
+
for out, res in zip(generated_outputs, results)
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
# capture iteration history
|
|
101
|
+
avg_score = sum(scores) / len(scores) if scores else 0.0
|
|
102
|
+
self.history_list.append(
|
|
103
|
+
IterationHistory(
|
|
104
|
+
prompt=prompt_text,
|
|
105
|
+
average_score=avg_score,
|
|
106
|
+
individual_results=results,
|
|
107
|
+
)
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
# Check early stopping
|
|
111
|
+
if self.early_stopping_checker:
|
|
112
|
+
if self.early_stopping_checker.should_stop(avg_score, len(batch)):
|
|
113
|
+
reason = self.early_stopping_checker.get_state()["stop_reason"]
|
|
114
|
+
logger.info(f"Early stopping triggered: {reason}")
|
|
115
|
+
raise EarlyStoppingException(reason)
|
|
116
|
+
|
|
117
|
+
trajectories = []
|
|
118
|
+
if capture_traces:
|
|
119
|
+
logger.info("Capturing traces.")
|
|
120
|
+
for i in range(len(batch)):
|
|
121
|
+
trajectories.append(
|
|
122
|
+
{
|
|
123
|
+
"inputs": batch[i],
|
|
124
|
+
"generated_output": generated_outputs[i],
|
|
125
|
+
"evaluation_result": results[i].model_dump(),
|
|
126
|
+
}
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
eval_end_time = time.time()
|
|
130
|
+
logger.info(f"Evaluation finished in {eval_end_time - eval_start_time:.2f}s.")
|
|
131
|
+
return EvaluationBatch(
|
|
132
|
+
outputs=outputs, scores=scores, trajectories=trajectories
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
def make_reflective_dataset(
|
|
136
|
+
self,
|
|
137
|
+
candidate: Dict[str, str],
|
|
138
|
+
eval_batch: EvaluationBatch,
|
|
139
|
+
components_to_update: List[str],
|
|
140
|
+
) -> Dict[str, List[Dict[str, Any]]]:
|
|
141
|
+
"""
|
|
142
|
+
Creates the dataset for GEPA's reflective LLM to analyze.
|
|
143
|
+
"""
|
|
144
|
+
logger.info("Creating reflective dataset.")
|
|
145
|
+
reflective_data = {comp: [] for comp in components_to_update}
|
|
146
|
+
|
|
147
|
+
if not eval_batch.trajectories:
|
|
148
|
+
logger.warning("No trajectories found to create reflective dataset.")
|
|
149
|
+
return reflective_data
|
|
150
|
+
|
|
151
|
+
logger.info(f"Processing {len(eval_batch.trajectories)} trajectories.")
|
|
152
|
+
for trajectory in eval_batch.trajectories:
|
|
153
|
+
result = trajectory.get("evaluation_result", {})
|
|
154
|
+
score = result.get("score", 0.0)
|
|
155
|
+
reason = result.get("reason", "No reason provided.")
|
|
156
|
+
|
|
157
|
+
if score >= 0.8:
|
|
158
|
+
feedback = f"This output was successful (score={score:.2f}). The reasoning for this score was: {reason}"
|
|
159
|
+
else:
|
|
160
|
+
feedback = f"This output performed poorly (score={score:.2f}). The key reason for the failure was: {reason}. The prompt needs to be improved to avoid this specific failure mode."
|
|
161
|
+
|
|
162
|
+
example = {
|
|
163
|
+
"Inputs": trajectory.get("inputs", {}),
|
|
164
|
+
"Generated Outputs": trajectory.get("generated_output", ""),
|
|
165
|
+
"Feedback": feedback,
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
for comp in components_to_update:
|
|
169
|
+
reflective_data[comp].append(example)
|
|
170
|
+
|
|
171
|
+
logger.info(
|
|
172
|
+
f"Reflective dataset created for components: {components_to_update}"
|
|
173
|
+
)
|
|
174
|
+
return reflective_data
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
class GEPAOptimizer(BaseOptimizer):
|
|
178
|
+
"""
|
|
179
|
+
An adapter that integrates the powerful GEPA evolutionary optimization
|
|
180
|
+
algorithm into the prompt-optimizer framework.
|
|
181
|
+
"""
|
|
182
|
+
|
|
183
|
+
def __init__(self, reflection_model: str, generator_model: str = "gpt-4o-mini"):
|
|
184
|
+
"""
|
|
185
|
+
Initializes the GEPA Optimizer wrapper.
|
|
186
|
+
|
|
187
|
+
Args:
|
|
188
|
+
reflection_model (str): The name of a powerful LLM (e.g., "gpt-4-turbo")
|
|
189
|
+
that GEPA will use for its reflection and mutation steps.
|
|
190
|
+
generator_model (str): The name of the model that will be used by the
|
|
191
|
+
prompts being optimized (the "task language model").
|
|
192
|
+
"""
|
|
193
|
+
self.reflection_model = reflection_model
|
|
194
|
+
self.generator_model = generator_model
|
|
195
|
+
logger.info(
|
|
196
|
+
f"Initialized with reflection_model: {reflection_model}, generator_model: {generator_model}"
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
def optimize(
|
|
200
|
+
self,
|
|
201
|
+
evaluator: Evaluator,
|
|
202
|
+
data_mapper: BasicDataMapper,
|
|
203
|
+
dataset: List[Dict[str, Any]],
|
|
204
|
+
initial_prompts: List[str],
|
|
205
|
+
max_metric_calls: Optional[int] = 150,
|
|
206
|
+
early_stopping: Optional[EarlyStoppingConfig] = None,
|
|
207
|
+
) -> OptimizationResult:
|
|
208
|
+
opt_start_time = time.time()
|
|
209
|
+
logger.info("--- Starting GEPA Prompt Optimization ---")
|
|
210
|
+
logger.info(f"Dataset size: {len(dataset)}")
|
|
211
|
+
logger.info(f"Initial prompts: {initial_prompts}")
|
|
212
|
+
logger.info(f"Max metric calls: {max_metric_calls}")
|
|
213
|
+
|
|
214
|
+
# Initialize early stopping checker
|
|
215
|
+
checker = None
|
|
216
|
+
if early_stopping and early_stopping.is_enabled():
|
|
217
|
+
checker = EarlyStoppingChecker(early_stopping)
|
|
218
|
+
logger.info(f"Early stopping enabled: {early_stopping}")
|
|
219
|
+
|
|
220
|
+
if not initial_prompts:
|
|
221
|
+
raise ValueError("Initial prompts list cannot be empty for GEPAOptimizer.")
|
|
222
|
+
history: List[IterationHistory] = []
|
|
223
|
+
# 1. Create the internal adapter that bridges our framework to GEPA
|
|
224
|
+
logger.info("Creating internal GEPA adapter...")
|
|
225
|
+
adapter = _InternalGEPAAdapter(
|
|
226
|
+
generator_model=self.generator_model,
|
|
227
|
+
evaluator=evaluator,
|
|
228
|
+
data_mapper=data_mapper,
|
|
229
|
+
history_list=history,
|
|
230
|
+
early_stopping_checker=checker,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
# 2. Prepare the inputs for gepa.optimize
|
|
234
|
+
seed_candidate = {"prompt": initial_prompts[0]}
|
|
235
|
+
logger.info(f"Seed candidate for GEPA: {seed_candidate}")
|
|
236
|
+
|
|
237
|
+
# 3. Call the external GEPA library's optimize function
|
|
238
|
+
logger.info("Calling gepa.optimize...")
|
|
239
|
+
gepa_start_time = time.time()
|
|
240
|
+
|
|
241
|
+
try:
|
|
242
|
+
gepa_result = gepa.optimize(
|
|
243
|
+
seed_candidate=seed_candidate,
|
|
244
|
+
trainset=dataset,
|
|
245
|
+
valset=dataset,
|
|
246
|
+
adapter=adapter,
|
|
247
|
+
reflection_lm=self.reflection_model,
|
|
248
|
+
max_metric_calls=max_metric_calls,
|
|
249
|
+
display_progress_bar=True,
|
|
250
|
+
)
|
|
251
|
+
gepa_end_time = time.time()
|
|
252
|
+
logger.info(
|
|
253
|
+
f"gepa.optimize finished in {gepa_end_time - gepa_start_time:.2f}s."
|
|
254
|
+
)
|
|
255
|
+
logger.info(
|
|
256
|
+
f"GEPA result best score: {gepa_result.val_aggregate_scores[gepa_result.best_idx]}"
|
|
257
|
+
)
|
|
258
|
+
logger.info(f"GEPA best candidate: {gepa_result.best_candidate}")
|
|
259
|
+
|
|
260
|
+
logger.info(f"Captured {len(history)} iterations in history.")
|
|
261
|
+
# 4. Translate GEPA's result back into our framework's standard format
|
|
262
|
+
logger.info("Translating GEPA result to OptimizationResult...")
|
|
263
|
+
|
|
264
|
+
final_best_generator = LiteLLMGenerator(
|
|
265
|
+
model=self.generator_model,
|
|
266
|
+
prompt_template=gepa_result.best_candidate.get("prompt", ""),
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
# Build result with early stopping metadata
|
|
270
|
+
result = OptimizationResult(
|
|
271
|
+
best_generator=final_best_generator,
|
|
272
|
+
history=history,
|
|
273
|
+
final_score=gepa_result.val_aggregate_scores[gepa_result.best_idx],
|
|
274
|
+
early_stopped=False,
|
|
275
|
+
stop_reason=None,
|
|
276
|
+
total_iterations=len(history),
|
|
277
|
+
total_evaluations=(
|
|
278
|
+
checker.get_state()["total_evaluations"]
|
|
279
|
+
if checker
|
|
280
|
+
else sum(len(h.individual_results) for h in history)
|
|
281
|
+
),
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
except EarlyStoppingException as e:
|
|
285
|
+
gepa_end_time = time.time()
|
|
286
|
+
logger.info(
|
|
287
|
+
f"GEPA stopped early after {gepa_end_time - gepa_start_time:.2f}s: {e.reason}"
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
# Use best from history
|
|
291
|
+
if not history:
|
|
292
|
+
raise RuntimeError(
|
|
293
|
+
"Early stopping triggered before any evaluations completed"
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
best_history = max(history, key=lambda h: h.average_score)
|
|
297
|
+
final_best_generator = LiteLLMGenerator(
|
|
298
|
+
model=self.generator_model,
|
|
299
|
+
prompt_template=best_history.prompt,
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
result = OptimizationResult(
|
|
303
|
+
best_generator=final_best_generator,
|
|
304
|
+
history=history,
|
|
305
|
+
final_score=best_history.average_score,
|
|
306
|
+
early_stopped=True,
|
|
307
|
+
stop_reason=e.reason,
|
|
308
|
+
total_iterations=len(history),
|
|
309
|
+
total_evaluations=(
|
|
310
|
+
checker.get_state()["total_evaluations"]
|
|
311
|
+
if checker
|
|
312
|
+
else sum(len(h.individual_results) for h in history)
|
|
313
|
+
),
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
opt_end_time = time.time()
|
|
317
|
+
logger.info(
|
|
318
|
+
f"--- GEPA Prompt Optimization finished in {opt_end_time - opt_start_time:.2f}s ---"
|
|
319
|
+
)
|
|
320
|
+
logger.info(f"Final best score: {result.final_score}")
|
|
321
|
+
|
|
322
|
+
return result
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import random
|
|
3
|
+
from typing import Any, Dict, List, Optional
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, Field, ValidationError
|
|
6
|
+
|
|
7
|
+
from ..base.base_optimizer import BaseOptimizer
|
|
8
|
+
from ..datamappers.basic_mapper import BasicDataMapper
|
|
9
|
+
from ..base.evaluator import Evaluator
|
|
10
|
+
from ..generators.litellm import LiteLLMGenerator
|
|
11
|
+
from ..types import IterationHistory, OptimizationResult
|
|
12
|
+
from ..utils.early_stopping import EarlyStoppingConfig, EarlyStoppingChecker
|
|
13
|
+
import logging
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
# ==============================================================================
|
|
17
|
+
# Prompts and Pydantic Models for the Teacher LLM (Meta-Model)
|
|
18
|
+
# ==============================================================================
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
META_PROMPT_TEMPLATE = """
|
|
22
|
+
You are a world-class expert in prompt engineering. Your task is to diagnose and optimize a given prompt based on its performance on a set of test cases.
|
|
23
|
+
|
|
24
|
+
### Current Prompt
|
|
25
|
+
The following is the current prompt being evaluated:
|
|
26
|
+
---
|
|
27
|
+
{current_prompt}
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
### Previous Failed Attempts
|
|
31
|
+
You have already tried the following prompts, but they performed worse than the current one. Analyze why they failed to avoid repeating mistakes.
|
|
32
|
+
---
|
|
33
|
+
{other_attempts}
|
|
34
|
+
---
|
|
35
|
+
|
|
36
|
+
### Performance Data
|
|
37
|
+
The current prompt was run on a set of examples, and here are the results. Pay close attention to the examples with low scores.
|
|
38
|
+
---
|
|
39
|
+
{annotated_results}
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
### Task Description
|
|
43
|
+
{task_description}
|
|
44
|
+
|
|
45
|
+
### Your Task
|
|
46
|
+
Think step-by-step to generate an improved prompt:
|
|
47
|
+
1. **Analyze Failures:** Deeply analyze the failing examples. What patterns do you see? Is the prompt too vague, too restrictive, or missing key instructions?
|
|
48
|
+
2. **Formulate a Hypothesis:** Based on your analysis, state a clear hypothesis for how to improve the prompt. For example, "My hypothesis is that adding a chain-of-thought instruction will improve reasoning on multi-step problems."
|
|
49
|
+
3. **Generate Improved Prompt:** Rewrite the *entire* prompt, implementing your hypothesis. The new prompt should be a complete replacement for the current one.
|
|
50
|
+
|
|
51
|
+
Return ONLY a valid JSON object with two keys: "hypothesis" (your string hypothesis) and "improved_prompt" (the complete new prompt string).
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class MetaPromptOutput(BaseModel):
|
|
56
|
+
hypothesis: str = Field(
|
|
57
|
+
description="The hypothesis for why the new prompt will be better."
|
|
58
|
+
)
|
|
59
|
+
improved_prompt: str = Field(description="The complete, new, improved prompt.")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# ==============================================================================
|
|
63
|
+
# The MetaPrompt Optimizer Class
|
|
64
|
+
# ==============================================================================
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class MetaPromptOptimizer(BaseOptimizer):
|
|
68
|
+
"""
|
|
69
|
+
Optimizes a prompt by using a powerful "teacher" LLM to analyze its
|
|
70
|
+
performance and rewrite it. This is inspired by the `promptim` library.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
def __init__(
|
|
74
|
+
self,
|
|
75
|
+
teacher_generator: LiteLLMGenerator,
|
|
76
|
+
task_model: Optional[str] = None,
|
|
77
|
+
):
|
|
78
|
+
"""
|
|
79
|
+
Initializes the MetaPrompt Optimizer.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
teacher_generator: A powerful generator (e.g., GPT-4o, Claude 3 Opus)
|
|
83
|
+
used to analyze performance and generate new prompts.
|
|
84
|
+
task_model: Model used to run candidate prompts while scoring them.
|
|
85
|
+
Defaults to the teacher generator's model.
|
|
86
|
+
"""
|
|
87
|
+
self.teacher = teacher_generator
|
|
88
|
+
self.task_model = task_model
|
|
89
|
+
|
|
90
|
+
def optimize(
|
|
91
|
+
self,
|
|
92
|
+
evaluator: Evaluator,
|
|
93
|
+
data_mapper: BasicDataMapper,
|
|
94
|
+
dataset: List[Dict[str, Any]],
|
|
95
|
+
initial_prompts: List[str],
|
|
96
|
+
task_description: str = "I want to improve my prompt.",
|
|
97
|
+
num_rounds: Optional[int] = 5,
|
|
98
|
+
eval_subset_size: Optional[int] = 40,
|
|
99
|
+
early_stopping: Optional[EarlyStoppingConfig] = None,
|
|
100
|
+
) -> OptimizationResult:
|
|
101
|
+
logger.info("--- Starting Meta-Prompt Optimization ---")
|
|
102
|
+
|
|
103
|
+
# Initialize early stopping checker
|
|
104
|
+
checker = None
|
|
105
|
+
if early_stopping and early_stopping.is_enabled():
|
|
106
|
+
checker = EarlyStoppingChecker(early_stopping)
|
|
107
|
+
logger.info(f"Early stopping enabled: {early_stopping}")
|
|
108
|
+
|
|
109
|
+
if not initial_prompts:
|
|
110
|
+
raise ValueError("Initial prompts list cannot be empty.")
|
|
111
|
+
|
|
112
|
+
current_prompt = initial_prompts[0]
|
|
113
|
+
best_prompt = current_prompt
|
|
114
|
+
best_score = -1.0
|
|
115
|
+
history: List[IterationHistory] = []
|
|
116
|
+
previous_attempts = set()
|
|
117
|
+
|
|
118
|
+
for round_num in range(num_rounds):
|
|
119
|
+
logger.info(
|
|
120
|
+
f"\n--- Starting Optimization Round {round_num + 1}/{num_rounds} ---"
|
|
121
|
+
)
|
|
122
|
+
logger.info(f"Current best prompt:\n{current_prompt}")
|
|
123
|
+
|
|
124
|
+
# 1. Evaluate the current prompt on a subset of data
|
|
125
|
+
eval_subset = random.sample(dataset, min(len(dataset), eval_subset_size))
|
|
126
|
+
iteration_history = self._score_prompt(
|
|
127
|
+
current_prompt, evaluator, data_mapper, eval_subset
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if not iteration_history:
|
|
131
|
+
logger.warning("Evaluation of current prompt failed. Skipping round.")
|
|
132
|
+
continue
|
|
133
|
+
|
|
134
|
+
history.append(iteration_history)
|
|
135
|
+
current_score = iteration_history.average_score
|
|
136
|
+
|
|
137
|
+
if current_score > best_score:
|
|
138
|
+
best_score = current_score
|
|
139
|
+
best_prompt = current_prompt
|
|
140
|
+
logger.info(f"New best score found: {best_score:.4f}")
|
|
141
|
+
|
|
142
|
+
# Check early stopping
|
|
143
|
+
if checker:
|
|
144
|
+
num_evals = len(eval_subset)
|
|
145
|
+
if checker.should_stop(current_score, num_evals):
|
|
146
|
+
logger.info(
|
|
147
|
+
f"Early stopping triggered: {checker.get_state()['stop_reason']}"
|
|
148
|
+
)
|
|
149
|
+
break
|
|
150
|
+
|
|
151
|
+
# 2. Use the teacher model to generate a new, improved prompt
|
|
152
|
+
annotated_results_str = self._format_results(iteration_history, eval_subset)
|
|
153
|
+
|
|
154
|
+
# Format previous attempts for the meta-prompt
|
|
155
|
+
other_attempts_str = (
|
|
156
|
+
"\n---\n".join(list(previous_attempts)) if previous_attempts else "N/A"
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
meta_prompt = META_PROMPT_TEMPLATE.format(
|
|
160
|
+
current_prompt=current_prompt,
|
|
161
|
+
other_attempts=other_attempts_str,
|
|
162
|
+
annotated_results=annotated_results_str,
|
|
163
|
+
task_description=task_description,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
logger.debug("Generating new prompt with meta-prompt...")
|
|
167
|
+
new_prompt_json = self.teacher.generate(
|
|
168
|
+
prompt_vars={"prompt": meta_prompt},
|
|
169
|
+
response_format={"type": "json_object"},
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
try:
|
|
173
|
+
parsed_output = MetaPromptOutput.model_validate_json(new_prompt_json)
|
|
174
|
+
logger.info(f"Teacher's Hypothesis: {parsed_output.hypothesis}")
|
|
175
|
+
previous_attempts.add(current_prompt)
|
|
176
|
+
current_prompt = parsed_output.improved_prompt
|
|
177
|
+
except (ValidationError, json.JSONDecodeError) as e:
|
|
178
|
+
logger.error(
|
|
179
|
+
f"Failed to parse new prompt from teacher model, keeping current prompt. Error: {e}"
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
final_best_generator = LiteLLMGenerator(self.teacher.model_name, best_prompt)
|
|
183
|
+
|
|
184
|
+
# Build result with early stopping metadata
|
|
185
|
+
return OptimizationResult(
|
|
186
|
+
best_generator=final_best_generator,
|
|
187
|
+
history=history,
|
|
188
|
+
final_score=best_score,
|
|
189
|
+
early_stopped=checker.get_state()["stopped"] if checker else False,
|
|
190
|
+
stop_reason=checker.get_state()["stop_reason"] if checker else None,
|
|
191
|
+
total_iterations=len(history),
|
|
192
|
+
total_evaluations=(
|
|
193
|
+
checker.get_state()["total_evaluations"]
|
|
194
|
+
if checker
|
|
195
|
+
else sum(len(h.individual_results) for h in history)
|
|
196
|
+
),
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
def _score_prompt(
|
|
200
|
+
self,
|
|
201
|
+
prompt: str,
|
|
202
|
+
evaluator: Evaluator,
|
|
203
|
+
data_mapper: BasicDataMapper,
|
|
204
|
+
dataset: List[Dict[str, Any]],
|
|
205
|
+
) -> IterationHistory | None:
|
|
206
|
+
"""Scores a single prompt and returns its history."""
|
|
207
|
+
try:
|
|
208
|
+
temp_generator = LiteLLMGenerator(
|
|
209
|
+
self.task_model or self.teacher.model_name, prompt
|
|
210
|
+
)
|
|
211
|
+
generated_outputs = [
|
|
212
|
+
temp_generator.generate(example) for example in dataset
|
|
213
|
+
]
|
|
214
|
+
eval_inputs = [
|
|
215
|
+
data_mapper.map(gen_out, ex)
|
|
216
|
+
for gen_out, ex in zip(generated_outputs, dataset)
|
|
217
|
+
]
|
|
218
|
+
results = evaluator.evaluate(eval_inputs)
|
|
219
|
+
avg_score = (
|
|
220
|
+
sum(res.score for res in results) / len(results) if results else 0.0
|
|
221
|
+
)
|
|
222
|
+
return IterationHistory(
|
|
223
|
+
prompt=prompt, average_score=avg_score, individual_results=results
|
|
224
|
+
)
|
|
225
|
+
except Exception as e:
|
|
226
|
+
logger.error(f"Failed to score prompt: {e}")
|
|
227
|
+
return None
|
|
228
|
+
|
|
229
|
+
def _format_results(
|
|
230
|
+
self, iteration_history: IterationHistory, dataset: List[Dict[str, Any]]
|
|
231
|
+
) -> str:
|
|
232
|
+
"""Formats the evaluation results into a string for the meta-prompt."""
|
|
233
|
+
formatted_lines = []
|
|
234
|
+
for i, result in enumerate(iteration_history.individual_results):
|
|
235
|
+
example_input = dataset[i]
|
|
236
|
+
formatted_lines.append(f"Example {i + 1}:")
|
|
237
|
+
formatted_lines.append(
|
|
238
|
+
f" Input: {json.dumps(example_input, ensure_ascii=False)}"
|
|
239
|
+
)
|
|
240
|
+
formatted_lines.append(f" Score: {result.score:.2f}")
|
|
241
|
+
formatted_lines.append(f" Reason: {result.reason}")
|
|
242
|
+
formatted_lines.append("---")
|
|
243
|
+
return "\n".join(formatted_lines)
|