agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,417 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
import random
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any, Dict, List, Set, Optional
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, ValidationError
|
|
8
|
+
|
|
9
|
+
from ..base.base_optimizer import BaseOptimizer
|
|
10
|
+
from ..datamappers.basic_mapper import BasicDataMapper
|
|
11
|
+
from ..base.evaluator import Evaluator
|
|
12
|
+
from ..generators.litellm import LiteLLMGenerator
|
|
13
|
+
from ..types import IterationHistory, OptimizationResult
|
|
14
|
+
from ..utils.early_stopping import EarlyStoppingConfig, EarlyStoppingChecker
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
MUTATE_PROMPT = """
|
|
19
|
+
You are an expert in prompt engineering. You will be given a task description and different styles known as meta prompts. Your task is to generate {num_variations} diverse variations of the following instruction by adaptively mixing meta prompt while keeping similar semantic meaning.
|
|
20
|
+
|
|
21
|
+
[Task Description]: {task_description}
|
|
22
|
+
[Meta Prompts]: {meta_prompts}
|
|
23
|
+
[Prompt Instruction]: {prompt_instruction}
|
|
24
|
+
|
|
25
|
+
Return ONLY a valid JSON object with a single key "variations" containing a list of the new prompt strings.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
CRITIQUE_PROMPT = """
|
|
29
|
+
You are an expert prompt engineering analyst. My current prompt is:
|
|
30
|
+
---
|
|
31
|
+
{instruction}
|
|
32
|
+
---
|
|
33
|
+
This prompt performed poorly on the following examples:
|
|
34
|
+
---
|
|
35
|
+
{examples}
|
|
36
|
+
---
|
|
37
|
+
Provide a detailed critique explaining the potential reasons for failure.
|
|
38
|
+
|
|
39
|
+
Return ONLY a valid JSON object with a single key "variations" containing a list with ONE string: your critique.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
REFINE_PROMPT = """
|
|
43
|
+
You are an expert prompt engineer. My current prompt is:
|
|
44
|
+
---
|
|
45
|
+
{instruction}
|
|
46
|
+
---
|
|
47
|
+
It failed on these examples:
|
|
48
|
+
---
|
|
49
|
+
{examples}
|
|
50
|
+
---
|
|
51
|
+
Here is a critique of the prompt's weaknesses: "{critique}"
|
|
52
|
+
|
|
53
|
+
Based on this critique, write {steps_per_sample} different, improved versions of the prompt.
|
|
54
|
+
|
|
55
|
+
Return ONLY a valid JSON object with a single key "variations" containing a list of the new prompt strings.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Variations(BaseModel):
|
|
60
|
+
variations: List[str]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class PromptWizardOptimizer(BaseOptimizer):
|
|
64
|
+
"""
|
|
65
|
+
An adapter for the PromptWizard optimization algorithm, using a multi-stage
|
|
66
|
+
process of mutation, critique, and refinement for prompt instructions.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
def __init__(
|
|
70
|
+
self,
|
|
71
|
+
teacher_generator: LiteLLMGenerator,
|
|
72
|
+
mutate_rounds: int = 3,
|
|
73
|
+
refine_iterations: int = 2,
|
|
74
|
+
beam_size: int = 1,
|
|
75
|
+
task_model: Optional[str] = None,
|
|
76
|
+
):
|
|
77
|
+
self.teacher = teacher_generator
|
|
78
|
+
self.task_model = task_model
|
|
79
|
+
self.mutate_rounds = mutate_rounds
|
|
80
|
+
self.refine_iterations = refine_iterations
|
|
81
|
+
self.beam_size = beam_size
|
|
82
|
+
self.thinking_styles = THINKING_STYLES
|
|
83
|
+
logger.info("--- PromptWizard Optimizer Initialized ---")
|
|
84
|
+
logger.debug(
|
|
85
|
+
f"Initialized with: mutate_rounds={mutate_rounds}, "
|
|
86
|
+
f"refine_iterations={refine_iterations}, beam_size={beam_size}"
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
def optimize(
|
|
90
|
+
self,
|
|
91
|
+
evaluator: Evaluator,
|
|
92
|
+
data_mapper: BasicDataMapper,
|
|
93
|
+
dataset: List[Dict[str, Any]],
|
|
94
|
+
initial_prompts: List[str],
|
|
95
|
+
task_description: str = "No task description given.",
|
|
96
|
+
early_stopping: Optional[EarlyStoppingConfig] = None,
|
|
97
|
+
**kwargs: Any,
|
|
98
|
+
) -> OptimizationResult:
|
|
99
|
+
eval_subset_size = kwargs.get("eval_subset_size", 25)
|
|
100
|
+
logger.info("--- Starting PromptWizard Optimization ---")
|
|
101
|
+
logger.debug(f"Task: {task_description}")
|
|
102
|
+
logger.debug(f"Initial prompts count: {len(initial_prompts)}")
|
|
103
|
+
logger.debug(f"Dataset size: {len(dataset)}")
|
|
104
|
+
logger.debug(f"Evaluation subset size: {eval_subset_size}")
|
|
105
|
+
|
|
106
|
+
# Initialize early stopping checker
|
|
107
|
+
checker = None
|
|
108
|
+
if early_stopping and early_stopping.is_enabled():
|
|
109
|
+
checker = EarlyStoppingChecker(early_stopping)
|
|
110
|
+
logger.info(f"Early stopping enabled: {early_stopping}")
|
|
111
|
+
|
|
112
|
+
if not initial_prompts:
|
|
113
|
+
raise ValueError("Initial prompts list cannot be empty.")
|
|
114
|
+
|
|
115
|
+
current_best_instruction = initial_prompts[0]
|
|
116
|
+
history: List[IterationHistory] = []
|
|
117
|
+
logger.info(f"Initial best instruction: '{current_best_instruction[:100]}...'")
|
|
118
|
+
|
|
119
|
+
for i in range(self.refine_iterations):
|
|
120
|
+
logger.info(
|
|
121
|
+
f"\n--- Instruction Refinement Iteration {i + 1}/{self.refine_iterations} ---"
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# 1. Mutate
|
|
125
|
+
logger.info("Step 1: Mutating instruction...")
|
|
126
|
+
mutated_prompts = self._mutate_instruction(
|
|
127
|
+
current_best_instruction, task_description
|
|
128
|
+
)
|
|
129
|
+
candidate_pool = {current_best_instruction, *mutated_prompts}
|
|
130
|
+
logger.info(f"Generated {len(mutated_prompts)} unique new prompts.")
|
|
131
|
+
logger.debug(f"Candidate pool size: {len(candidate_pool)}")
|
|
132
|
+
|
|
133
|
+
# 2. Score
|
|
134
|
+
logger.info("Step 2: Scoring candidate prompts...")
|
|
135
|
+
eval_subset = random.sample(dataset, min(len(dataset), eval_subset_size))
|
|
136
|
+
logger.debug(f"Scoring against a subset of {len(eval_subset)} examples.")
|
|
137
|
+
iteration_history = self._score_candidates(
|
|
138
|
+
list(candidate_pool), evaluator, data_mapper, eval_subset
|
|
139
|
+
)
|
|
140
|
+
history.extend(iteration_history)
|
|
141
|
+
|
|
142
|
+
sorted_by_score = sorted(
|
|
143
|
+
iteration_history, key=lambda x: x.average_score, reverse=True
|
|
144
|
+
)
|
|
145
|
+
top_prompts_this_round = [
|
|
146
|
+
item.prompt for item in sorted_by_score[: self.beam_size]
|
|
147
|
+
]
|
|
148
|
+
logger.info(f"Top {self.beam_size} prompts selected for refinement.")
|
|
149
|
+
for idx, p in enumerate(top_prompts_this_round):
|
|
150
|
+
score = sorted_by_score[idx].average_score
|
|
151
|
+
logger.debug(f" - Prompt (Score: {score:.4f}): '{p[:100]}...'")
|
|
152
|
+
|
|
153
|
+
# Check early stopping
|
|
154
|
+
if checker:
|
|
155
|
+
best_round_score = sorted_by_score[0].average_score
|
|
156
|
+
num_evals = len(candidate_pool) * len(eval_subset)
|
|
157
|
+
if checker.should_stop(best_round_score, num_evals):
|
|
158
|
+
logger.info(
|
|
159
|
+
f"Early stopping triggered: {checker.get_state()['stop_reason']}"
|
|
160
|
+
)
|
|
161
|
+
break
|
|
162
|
+
|
|
163
|
+
# 3. Critique and Refine
|
|
164
|
+
logger.info("Step 3: Critiquing and refining top prompts...")
|
|
165
|
+
refined_prompts = set()
|
|
166
|
+
for prompt_to_refine in top_prompts_this_round:
|
|
167
|
+
errors = self._get_errors(
|
|
168
|
+
prompt_to_refine, evaluator, data_mapper, dataset
|
|
169
|
+
)
|
|
170
|
+
if errors:
|
|
171
|
+
logger.debug(
|
|
172
|
+
f"Found {len(errors)} errors for prompt: '{prompt_to_refine[:100]}...'"
|
|
173
|
+
)
|
|
174
|
+
refined = self._critique_and_refine(prompt_to_refine, errors)
|
|
175
|
+
if refined:
|
|
176
|
+
logger.debug("Successfully refined prompt.")
|
|
177
|
+
refined_prompts.add(refined)
|
|
178
|
+
else:
|
|
179
|
+
logger.debug(
|
|
180
|
+
f"No errors found for prompt, skipping refinement: '{prompt_to_refine[:100]}...'"
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
# Determine the best instruction for the next iteration
|
|
184
|
+
if refined_prompts:
|
|
185
|
+
logger.info(
|
|
186
|
+
f"Scoring {len(refined_prompts)} refined prompts to find new best."
|
|
187
|
+
)
|
|
188
|
+
final_candidates_this_round = {
|
|
189
|
+
current_best_instruction,
|
|
190
|
+
*refined_prompts,
|
|
191
|
+
}
|
|
192
|
+
final_history = self._score_candidates(
|
|
193
|
+
list(final_candidates_this_round),
|
|
194
|
+
evaluator,
|
|
195
|
+
data_mapper,
|
|
196
|
+
eval_subset,
|
|
197
|
+
)
|
|
198
|
+
history.extend(final_history)
|
|
199
|
+
current_best_instruction = sorted(
|
|
200
|
+
final_history, key=lambda x: x.average_score, reverse=True
|
|
201
|
+
)[0].prompt
|
|
202
|
+
else:
|
|
203
|
+
logger.info("No prompts were refined, carrying over previous best.")
|
|
204
|
+
current_best_instruction = top_prompts_this_round[0]
|
|
205
|
+
|
|
206
|
+
logger.info(
|
|
207
|
+
f"Best instruction after iteration {i + 1}: '{current_best_instruction[:100]}...'"
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
logger.info("--- PromptWizard Optimization Finished ---")
|
|
211
|
+
final_history = sorted(history, key=lambda x: x.average_score, reverse=True)
|
|
212
|
+
best_prompt = final_history[0].prompt
|
|
213
|
+
best_score = final_history[0].average_score
|
|
214
|
+
logger.info(f"Final best prompt (Score: {best_score:.4f}): '{best_prompt}'")
|
|
215
|
+
|
|
216
|
+
final_best_generator = LiteLLMGenerator(self.teacher.model_name, best_prompt)
|
|
217
|
+
|
|
218
|
+
# Build result with early stopping metadata
|
|
219
|
+
return OptimizationResult(
|
|
220
|
+
best_generator=final_best_generator,
|
|
221
|
+
history=history,
|
|
222
|
+
final_score=best_score,
|
|
223
|
+
early_stopped=checker.get_state()["stopped"] if checker else False,
|
|
224
|
+
stop_reason=checker.get_state()["stop_reason"] if checker else None,
|
|
225
|
+
total_iterations=len(history),
|
|
226
|
+
total_evaluations=(
|
|
227
|
+
checker.get_state()["total_evaluations"]
|
|
228
|
+
if checker
|
|
229
|
+
else sum(len(h.individual_results) for h in history)
|
|
230
|
+
),
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
def _mutate_instruction(
|
|
234
|
+
self,
|
|
235
|
+
base_instruction: str,
|
|
236
|
+
task_description: str,
|
|
237
|
+
) -> Set[str]:
|
|
238
|
+
logger.debug(
|
|
239
|
+
f"Entering mutation phase for instruction: '{base_instruction[:100]}...'"
|
|
240
|
+
)
|
|
241
|
+
all_variations = set()
|
|
242
|
+
temp_generator = LiteLLMGenerator(
|
|
243
|
+
self.task_model or self.teacher.model_name, "{prompt}"
|
|
244
|
+
)
|
|
245
|
+
for i in range(self.mutate_rounds):
|
|
246
|
+
logger.debug(f"Mutation round {i + 1}/{self.mutate_rounds}")
|
|
247
|
+
prompt = MUTATE_PROMPT.format(
|
|
248
|
+
num_variations=len(self.thinking_styles),
|
|
249
|
+
task_description=task_description,
|
|
250
|
+
meta_prompts="\n".join(self.thinking_styles),
|
|
251
|
+
prompt_instruction=base_instruction,
|
|
252
|
+
)
|
|
253
|
+
response_text = temp_generator.generate(
|
|
254
|
+
{"prompt": prompt}, response_format={"type": "json_object"}
|
|
255
|
+
)
|
|
256
|
+
variations = self._parse_variations_from_json(response_text)
|
|
257
|
+
logger.debug(f"Generated {len(variations)} variations in this round.")
|
|
258
|
+
all_variations.update(variations)
|
|
259
|
+
logger.debug(f"Total unique variations from mutation: {len(all_variations)}")
|
|
260
|
+
return all_variations
|
|
261
|
+
|
|
262
|
+
def _critique_and_refine(
|
|
263
|
+
self,
|
|
264
|
+
prompt: str,
|
|
265
|
+
errors: List[Dict[str, Any]],
|
|
266
|
+
) -> str | None:
|
|
267
|
+
logger.debug(f"Entering critique and refine for: '{prompt[:100]}...'")
|
|
268
|
+
error_str = json.dumps(errors, indent=2, ensure_ascii=False)
|
|
269
|
+
|
|
270
|
+
logger.debug("Generating critique...")
|
|
271
|
+
critique_prompt = CRITIQUE_PROMPT.format(instruction=prompt, examples=error_str)
|
|
272
|
+
critique_response = self.teacher.generate(
|
|
273
|
+
{"prompt": critique_prompt}, response_format={"type": "json_object"}
|
|
274
|
+
)
|
|
275
|
+
critiques = self._parse_variations_from_json(critique_response)
|
|
276
|
+
if not critiques:
|
|
277
|
+
logger.warning("Critique generation failed, skipping refinement.")
|
|
278
|
+
return None
|
|
279
|
+
critique_text = "\n".join([critique for critique in critiques])
|
|
280
|
+
logger.debug(f"Generated critique: '{critique_text[:100]}...'")
|
|
281
|
+
|
|
282
|
+
logger.debug("Refining prompt based on critique...")
|
|
283
|
+
refine_prompt = REFINE_PROMPT.format(
|
|
284
|
+
instruction=prompt,
|
|
285
|
+
examples=error_str,
|
|
286
|
+
critique=critique_text,
|
|
287
|
+
steps_per_sample=1,
|
|
288
|
+
)
|
|
289
|
+
refined_text = self.teacher.generate(
|
|
290
|
+
{"prompt": refine_prompt}, response_format={"type": "json_object"}
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
refined_prompts = self._parse_variations_from_json(refined_text)
|
|
294
|
+
if refined_prompts:
|
|
295
|
+
logger.debug(f"Refined prompt: '{refined_prompts[0][:100]}...'")
|
|
296
|
+
return refined_prompts[0]
|
|
297
|
+
else:
|
|
298
|
+
logger.warning("Refinement step produced no new prompts.")
|
|
299
|
+
return None
|
|
300
|
+
|
|
301
|
+
# Helper methods (shared with ProTeGi, kept here for encapsulation)
|
|
302
|
+
def _get_errors(
|
|
303
|
+
self,
|
|
304
|
+
prompt: str,
|
|
305
|
+
evaluator: Evaluator,
|
|
306
|
+
data_mapper: BasicDataMapper,
|
|
307
|
+
dataset: List[Dict[str, Any]],
|
|
308
|
+
sample_size: int = 10,
|
|
309
|
+
) -> List[Dict[str, Any]]:
|
|
310
|
+
logger.debug(f"Getting errors for prompt: '{prompt[:100]}...'")
|
|
311
|
+
subset = random.sample(dataset, min(len(dataset), sample_size))
|
|
312
|
+
temp_generator = LiteLLMGenerator(
|
|
313
|
+
self.task_model or self.teacher.model_name, prompt
|
|
314
|
+
)
|
|
315
|
+
generated_outputs = [temp_generator.generate(example) for example in subset]
|
|
316
|
+
eval_inputs = [
|
|
317
|
+
data_mapper.map(gen_out, ex)
|
|
318
|
+
for gen_out, ex in zip(generated_outputs, subset)
|
|
319
|
+
]
|
|
320
|
+
results = evaluator.evaluate(eval_inputs)
|
|
321
|
+
errors = [
|
|
322
|
+
{"inputs": subset[i], "output": generated_outputs[i], "score": res.score}
|
|
323
|
+
for i, res in enumerate(results)
|
|
324
|
+
if res.score < 0.5
|
|
325
|
+
]
|
|
326
|
+
logger.debug(f"Found {len(errors)} examples with score < 0.5.")
|
|
327
|
+
return errors
|
|
328
|
+
|
|
329
|
+
def _score_candidates(
|
|
330
|
+
self,
|
|
331
|
+
prompts: List[str],
|
|
332
|
+
evaluator: Evaluator,
|
|
333
|
+
data_mapper: BasicDataMapper,
|
|
334
|
+
dataset: List[Dict[str, Any]],
|
|
335
|
+
) -> List[IterationHistory]:
|
|
336
|
+
logger.debug(f"Scoring {len(prompts)} candidate prompts.")
|
|
337
|
+
histories = []
|
|
338
|
+
for i, prompt in enumerate(prompts):
|
|
339
|
+
temp_generator = LiteLLMGenerator(
|
|
340
|
+
self.task_model or self.teacher.model_name, prompt
|
|
341
|
+
)
|
|
342
|
+
generated_outputs = [
|
|
343
|
+
temp_generator.generate(example) for example in dataset
|
|
344
|
+
]
|
|
345
|
+
eval_inputs = [
|
|
346
|
+
data_mapper.map(gen_out, ex)
|
|
347
|
+
for gen_out, ex in zip(generated_outputs, dataset)
|
|
348
|
+
]
|
|
349
|
+
results = evaluator.evaluate(eval_inputs)
|
|
350
|
+
avg_score = (
|
|
351
|
+
sum(res.score for res in results) / len(results) if results else 0.0
|
|
352
|
+
)
|
|
353
|
+
logger.debug(
|
|
354
|
+
f" - Scored prompt {i + 1}/{len(prompts)} (Avg Score: {avg_score:.4f}): '{prompt[:100]}...'"
|
|
355
|
+
)
|
|
356
|
+
histories.append(
|
|
357
|
+
IterationHistory(
|
|
358
|
+
prompt=prompt, average_score=avg_score, individual_results=results
|
|
359
|
+
)
|
|
360
|
+
)
|
|
361
|
+
return histories
|
|
362
|
+
|
|
363
|
+
@staticmethod
|
|
364
|
+
def _parse_variations_from_json(text: str) -> List[str]:
|
|
365
|
+
text = text.strip()
|
|
366
|
+
|
|
367
|
+
# --- Stage 1: Try to parse the entire string as JSON ---
|
|
368
|
+
try:
|
|
369
|
+
data = json.loads(text)
|
|
370
|
+
return Variations.model_validate(data).variations
|
|
371
|
+
except (json.JSONDecodeError, ValidationError):
|
|
372
|
+
# This is expected if there's extra text, so we continue.
|
|
373
|
+
pass
|
|
374
|
+
|
|
375
|
+
# --- Stage 2: Look for a JSON markdown code block ---
|
|
376
|
+
try:
|
|
377
|
+
match = re.search(r"```json\s*(\{.*?\})\s*```", text, re.DOTALL)
|
|
378
|
+
if match:
|
|
379
|
+
json_str = match.group(1)
|
|
380
|
+
data = json.loads(json_str)
|
|
381
|
+
return Variations.model_validate(data).variations
|
|
382
|
+
except (json.JSONDecodeError, ValidationError):
|
|
383
|
+
pass
|
|
384
|
+
|
|
385
|
+
# --- Stage 3: Greedy fallback to find the first '{' and last '}' ---
|
|
386
|
+
try:
|
|
387
|
+
start_index = text.find("{")
|
|
388
|
+
end_index = text.rfind("}")
|
|
389
|
+
if start_index != -1 and end_index != -1 and end_index > start_index:
|
|
390
|
+
json_str = text[start_index : end_index + 1]
|
|
391
|
+
data = json.loads(json_str)
|
|
392
|
+
return Variations.model_validate(data).variations
|
|
393
|
+
except (json.JSONDecodeError, ValidationError) as e:
|
|
394
|
+
# If all parsing attempts fail, log the error and return empty.
|
|
395
|
+
logging.error(
|
|
396
|
+
f"Failed to parse teacher model JSON response after all fallbacks: {e}"
|
|
397
|
+
)
|
|
398
|
+
logging.debug(f"Raw problematic output that failed parsing:\n{text}")
|
|
399
|
+
return []
|
|
400
|
+
|
|
401
|
+
# If no JSON object is found at all
|
|
402
|
+
logging.warning("Could not find any JSON in the teacher's response.")
|
|
403
|
+
logging.debug(f"Raw response with no JSON:\n{text}")
|
|
404
|
+
return []
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
# Static list of thinking styles from PromptWizard's config
|
|
408
|
+
THINKING_STYLES = [
|
|
409
|
+
"How could I devise an experiment to help solve that problem?",
|
|
410
|
+
"Make a list of ideas for solving this problem, and apply them one by one.",
|
|
411
|
+
"How can I simplify the problem so that it is easier to solve?",
|
|
412
|
+
"What are the key assumptions underlying this problem?",
|
|
413
|
+
"Critical Thinking: Analyze the problem from different perspectives, questioning assumptions.",
|
|
414
|
+
"Try creative thinking, generate innovative and out-of-the-box ideas.",
|
|
415
|
+
"Use systems thinking: Consider the problem as part of a larger interconnected system.",
|
|
416
|
+
"Let's think step by step.",
|
|
417
|
+
]
|