agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
"""One conversation, one folder.
|
|
2
|
+
|
|
3
|
+
Everything about testing one agent lives in a single directory: what the agent is, the world
|
|
4
|
+
built for it, the scenarios written against that world, what happened when they ran, and the
|
|
5
|
+
conversation that produced all of it.
|
|
6
|
+
|
|
7
|
+
That is the whole state model. There is nothing held in memory that is not also on disk, so
|
|
8
|
+
closing the page, restarting the server or coming back tomorrow all resume the same way — by
|
|
9
|
+
reading the folder. A session that only existed in a process would be a session you could lose
|
|
10
|
+
by refreshing.
|
|
11
|
+
|
|
12
|
+
artifacts/sessions/<id>/
|
|
13
|
+
session.json what this is: the agent, where its source lives, when it started
|
|
14
|
+
chat.jsonl the conversation, one message per line
|
|
15
|
+
contract.json stage 1
|
|
16
|
+
world.sqlite stage 2, with handlers/, simulator_prompt.md, sub_goals.json
|
|
17
|
+
scenarios/<name>/ stage 3, one folder each
|
|
18
|
+
runs.json stage 4
|
|
19
|
+
|
|
20
|
+
The id is readable and unique: the agent's name with a short suffix, so two attempts at the same
|
|
21
|
+
agent are two sessions rather than one overwriting the other.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import json
|
|
27
|
+
import re
|
|
28
|
+
import secrets
|
|
29
|
+
import shutil
|
|
30
|
+
import time
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
from .config import ARTIFACTS_ROOT
|
|
36
|
+
|
|
37
|
+
SESSIONS = ARTIFACTS_ROOT / "sessions"
|
|
38
|
+
META = "session.json"
|
|
39
|
+
CHAT = "chat.jsonl"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _slug(text: str) -> str:
|
|
43
|
+
cleaned = re.sub(r"[^a-z0-9]+", "-", (text or "session").lower()).strip("-")
|
|
44
|
+
return cleaned[:32] or "session"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def root(base: Path | None = None) -> Path:
|
|
48
|
+
return Path(base) if base else SESSIONS
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def new_id(agent: str = "", base: Path | None = None) -> str:
|
|
52
|
+
"""A readable, unique id. Two goes at the same agent are two sessions, not one clobbered."""
|
|
53
|
+
stem = _slug(agent)
|
|
54
|
+
while True:
|
|
55
|
+
candidate = f"{stem}-{secrets.token_hex(3)}"
|
|
56
|
+
if not (root(base) / candidate).exists():
|
|
57
|
+
return candidate
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class Session:
|
|
62
|
+
"""One conversation's folder, and what is in it."""
|
|
63
|
+
|
|
64
|
+
id: str
|
|
65
|
+
path: Path
|
|
66
|
+
agent: str = ""
|
|
67
|
+
source: str = ""
|
|
68
|
+
kind: str = "repo"
|
|
69
|
+
created: float = 0.0
|
|
70
|
+
updated: float = 0.0
|
|
71
|
+
stage: str = ""
|
|
72
|
+
title: str = ""
|
|
73
|
+
|
|
74
|
+
def meta(self) -> dict[str, Any]:
|
|
75
|
+
return {
|
|
76
|
+
"id": self.id,
|
|
77
|
+
"agent": self.agent,
|
|
78
|
+
"source": self.source,
|
|
79
|
+
"kind": self.kind,
|
|
80
|
+
"created": self.created,
|
|
81
|
+
"updated": self.updated,
|
|
82
|
+
"stage": self.stage,
|
|
83
|
+
"title": self.title,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
def has(self) -> dict[str, Any]:
|
|
87
|
+
"""What this session has actually produced, read from the folder rather than remembered.
|
|
88
|
+
|
|
89
|
+
Asking the folder means the answer survives a restart, and it cannot drift from what is
|
|
90
|
+
really there — which is what makes reopening a session trustworthy.
|
|
91
|
+
"""
|
|
92
|
+
from .catalogue import load_catalogue
|
|
93
|
+
from .folder import read_all
|
|
94
|
+
from .world.snapshot import saved as world_saved
|
|
95
|
+
|
|
96
|
+
scenarios = read_all(self.path) if self.path.exists() else []
|
|
97
|
+
runs = _runs(self.path)
|
|
98
|
+
return {
|
|
99
|
+
"contract": (self.path / "contract.json").exists(),
|
|
100
|
+
"world": world_saved(self.path),
|
|
101
|
+
"simulator_prompt": (self.path / "simulator_prompt.md").exists(),
|
|
102
|
+
"sub_goals": len(load_catalogue(self.path).sub_goals)
|
|
103
|
+
if self.path.exists()
|
|
104
|
+
else 0,
|
|
105
|
+
"scenarios": len(scenarios),
|
|
106
|
+
"validated": None, # filled in by whoever wants to pay for proving them
|
|
107
|
+
"runs": len(runs),
|
|
108
|
+
"runs_passed": sum(1 for one in runs if one.get("passed")),
|
|
109
|
+
"messages": count_messages(self.path),
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _runs(path: Path) -> list[dict[str, Any]]:
|
|
114
|
+
# A completed timestamped simulation is the authoritative run record. ``runs.json`` is the
|
|
115
|
+
# compatibility file written by the older chat/local runner and may still be present after a
|
|
116
|
+
# voice simulation. Looking at it first made the Runs tab keep showing a discarded local
|
|
117
|
+
# attempt while the Simulations page correctly showed the newer WebRTC calls.
|
|
118
|
+
modern = sorted((path / "runs").glob("run-*/run.json"))
|
|
119
|
+
if modern:
|
|
120
|
+
from .run.simulation import read_run
|
|
121
|
+
|
|
122
|
+
newest = modern[-1]
|
|
123
|
+
return list(read_run(path, newest.parent.name).get("scenarios") or [])
|
|
124
|
+
|
|
125
|
+
found = path / "runs.json"
|
|
126
|
+
if not found.exists():
|
|
127
|
+
# The suite runner writes one rich timestamped folder per suite. The RL Environment's
|
|
128
|
+
# Runs tab predates that layout and reads this flattened view; without the bridge, the
|
|
129
|
+
# Simulations page showed the call while the adjacent Runs tab and stage badge said zero.
|
|
130
|
+
# Native WebRTC campaigns preserve richer per-case artifacts in timestamped folders.
|
|
131
|
+
# Show the newest completed campaign in the same Runs tab instead of making a
|
|
132
|
+
# successful external call campaign look as though nothing has ever run.
|
|
133
|
+
batches = sorted((path / "webrtc-runs").glob("run_*/results.json"))
|
|
134
|
+
if not batches:
|
|
135
|
+
return []
|
|
136
|
+
found = batches[-1]
|
|
137
|
+
try:
|
|
138
|
+
loaded = json.loads(found.read_text(encoding="utf-8"))
|
|
139
|
+
if not isinstance(loaded, list):
|
|
140
|
+
return []
|
|
141
|
+
if found.name == "runs.json":
|
|
142
|
+
return loaded
|
|
143
|
+
return [_webrtc_run(one) for one in loaded if isinstance(one, dict)]
|
|
144
|
+
except json.JSONDecodeError:
|
|
145
|
+
return []
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _webrtc_run(record: dict[str, Any]) -> dict[str, Any]:
|
|
149
|
+
"""Present a native WebRTC result in the live-run shape the UI already renders."""
|
|
150
|
+
calls = []
|
|
151
|
+
for call in record.get("tool_calls") or []:
|
|
152
|
+
if not isinstance(call, dict):
|
|
153
|
+
continue
|
|
154
|
+
arguments = json.dumps(
|
|
155
|
+
call.get("arguments") or {}, ensure_ascii=False, sort_keys=True
|
|
156
|
+
)
|
|
157
|
+
outcome = "ok" if call.get("ok") else "crashed"
|
|
158
|
+
calls.append(f"{call.get('name', 'unknown')}({arguments}) -> {outcome}")
|
|
159
|
+
problems = []
|
|
160
|
+
status = str(record.get("voice_status") or "")
|
|
161
|
+
if status and status != "completed":
|
|
162
|
+
problems.append(f"WebRTC call ended with voice status: {status}")
|
|
163
|
+
if record.get("error"):
|
|
164
|
+
problems.append(str(record["error"]))
|
|
165
|
+
return {
|
|
166
|
+
"scenario": record.get("scenario") or "unknown",
|
|
167
|
+
"passed": bool(record.get("passed")),
|
|
168
|
+
"met": int(record.get("deterministic_met") or 0),
|
|
169
|
+
"of": int(record.get("deterministic_of") or 0),
|
|
170
|
+
"settled": record.get("settled") or [],
|
|
171
|
+
"judged": record.get("judged") or [],
|
|
172
|
+
"calls": calls,
|
|
173
|
+
"problems": problems,
|
|
174
|
+
"transcript": record.get("transcript") or "",
|
|
175
|
+
"ended": status,
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def create(
|
|
180
|
+
agent: str = "", source: str = "", kind: str = "repo", base: Path | None = None
|
|
181
|
+
) -> Session:
|
|
182
|
+
"""Start a new conversation, with its own folder."""
|
|
183
|
+
identifier = new_id(agent, base)
|
|
184
|
+
path = root(base) / identifier
|
|
185
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
186
|
+
now = time.time()
|
|
187
|
+
session = Session(
|
|
188
|
+
id=identifier,
|
|
189
|
+
path=path,
|
|
190
|
+
agent=agent,
|
|
191
|
+
source=source,
|
|
192
|
+
kind=kind,
|
|
193
|
+
created=now,
|
|
194
|
+
updated=now,
|
|
195
|
+
stage="reception",
|
|
196
|
+
title=agent or "new session",
|
|
197
|
+
)
|
|
198
|
+
save(session)
|
|
199
|
+
return session
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def save(session: Session) -> None:
|
|
203
|
+
session.updated = time.time()
|
|
204
|
+
session.path.mkdir(parents=True, exist_ok=True)
|
|
205
|
+
(session.path / META).write_text(
|
|
206
|
+
json.dumps(session.meta(), indent=2, ensure_ascii=False), encoding="utf-8"
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def load(identifier: str, base: Path | None = None) -> Session | None:
|
|
211
|
+
path = root(base) / identifier
|
|
212
|
+
if not path.is_dir():
|
|
213
|
+
return None
|
|
214
|
+
body: dict[str, Any] = {}
|
|
215
|
+
found = path / META
|
|
216
|
+
if found.exists():
|
|
217
|
+
try:
|
|
218
|
+
body = json.loads(found.read_text(encoding="utf-8"))
|
|
219
|
+
except json.JSONDecodeError:
|
|
220
|
+
body = {}
|
|
221
|
+
return Session(
|
|
222
|
+
id=identifier,
|
|
223
|
+
path=path,
|
|
224
|
+
agent=str(body.get("agent") or ""),
|
|
225
|
+
source=str(body.get("source") or ""),
|
|
226
|
+
kind=str(body.get("kind") or "repo"),
|
|
227
|
+
created=float(body.get("created") or path.stat().st_ctime),
|
|
228
|
+
updated=float(body.get("updated") or path.stat().st_mtime),
|
|
229
|
+
stage=str(body.get("stage") or ""),
|
|
230
|
+
title=str(body.get("title") or identifier),
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def every(base: Path | None = None) -> list[Session]:
|
|
235
|
+
"""Every session, newest first."""
|
|
236
|
+
here = root(base)
|
|
237
|
+
if not here.exists():
|
|
238
|
+
return []
|
|
239
|
+
found = [load(one.name, base) for one in here.iterdir() if one.is_dir()]
|
|
240
|
+
return sorted((one for one in found if one), key=lambda s: s.updated, reverse=True)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def remove(identifier: str, base: Path | None = None) -> bool:
|
|
244
|
+
"""Delete a session and everything in it.
|
|
245
|
+
|
|
246
|
+
Deliberately narrow: it will only remove a directory that sits directly inside the sessions
|
|
247
|
+
root and holds a session file, so a mistyped id can never take anything else with it.
|
|
248
|
+
"""
|
|
249
|
+
here = (root(base) / identifier).resolve()
|
|
250
|
+
parent = root(base).resolve()
|
|
251
|
+
if here.parent != parent or not here.is_dir():
|
|
252
|
+
return False
|
|
253
|
+
if not (here / META).exists():
|
|
254
|
+
return False
|
|
255
|
+
shutil.rmtree(here)
|
|
256
|
+
return True
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
# -- the conversation itself --------------------------------------------------------
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
@dataclass
|
|
263
|
+
class Message:
|
|
264
|
+
"""One thing said, by either side."""
|
|
265
|
+
|
|
266
|
+
role: str # "you" or "harness"
|
|
267
|
+
text: str = ""
|
|
268
|
+
stage: str = ""
|
|
269
|
+
at: float = 0.0
|
|
270
|
+
# What the harness did while answering, so a reopened conversation shows the work and not
|
|
271
|
+
# only the conclusion.
|
|
272
|
+
tools: list[dict[str, Any]] = field(default_factory=list)
|
|
273
|
+
|
|
274
|
+
def body(self) -> dict[str, Any]:
|
|
275
|
+
return {
|
|
276
|
+
"role": self.role,
|
|
277
|
+
"text": self.text,
|
|
278
|
+
"stage": self.stage,
|
|
279
|
+
"at": self.at or time.time(),
|
|
280
|
+
"tools": self.tools,
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def remember(path: Path, message: Message) -> None:
|
|
285
|
+
"""Append one message to this session's conversation."""
|
|
286
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
287
|
+
with (path / CHAT).open("a", encoding="utf-8") as file:
|
|
288
|
+
file.write(json.dumps(message.body(), ensure_ascii=False) + "\n")
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def history(path: Path) -> list[dict[str, Any]]:
|
|
292
|
+
"""The whole conversation, in order.
|
|
293
|
+
|
|
294
|
+
A line that will not parse is skipped rather than taking the rest with it: a half-written
|
|
295
|
+
line at the end is the ordinary result of a process being killed mid-write, and losing the
|
|
296
|
+
conversation because of it would be absurd.
|
|
297
|
+
"""
|
|
298
|
+
found = Path(path) / CHAT
|
|
299
|
+
if not found.exists():
|
|
300
|
+
return []
|
|
301
|
+
messages: list[dict[str, Any]] = []
|
|
302
|
+
for line in found.read_text(encoding="utf-8").splitlines():
|
|
303
|
+
line = line.strip()
|
|
304
|
+
if not line:
|
|
305
|
+
continue
|
|
306
|
+
try:
|
|
307
|
+
messages.append(json.loads(line))
|
|
308
|
+
except json.JSONDecodeError:
|
|
309
|
+
continue
|
|
310
|
+
return messages
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def count_messages(path: Path) -> int:
|
|
314
|
+
return len(history(path))
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def environments(base: Path | None = None) -> list[dict[str, Any]]:
|
|
318
|
+
"""Every session with at least a contract, newest first, one table row each.
|
|
319
|
+
|
|
320
|
+
Read off each folder without adopting it, so listing environments can never
|
|
321
|
+
move the open session. A session appears as soon as its contract is written
|
|
322
|
+
— an hour-long build that shows nothing until its last artifact reads as an
|
|
323
|
+
empty product — and ``state`` says whether the world is there yet. Runs are
|
|
324
|
+
simulation runs; the legacy chat runs in ``runs.json`` are a different
|
|
325
|
+
thing and would double-count a session's work.
|
|
326
|
+
"""
|
|
327
|
+
from .run.simulation import every_run
|
|
328
|
+
|
|
329
|
+
found: list[dict[str, Any]] = []
|
|
330
|
+
for one in every(base):
|
|
331
|
+
held = one.has()
|
|
332
|
+
if not held.get("contract"):
|
|
333
|
+
continue
|
|
334
|
+
contract = _read_json(one.path / "contract.json")
|
|
335
|
+
manifest = _read_json(one.path / "manifest.json")
|
|
336
|
+
runs = every_run(one.path)
|
|
337
|
+
found.append(
|
|
338
|
+
{
|
|
339
|
+
"session_id": one.id,
|
|
340
|
+
"state": "ready" if held.get("world") else "building",
|
|
341
|
+
"agent": one.agent or contract.get("agent", ""),
|
|
342
|
+
"title": one.title or one.agent or one.id,
|
|
343
|
+
"one_liner": contract.get("one_liner", ""),
|
|
344
|
+
"created": one.created,
|
|
345
|
+
"updated": one.updated,
|
|
346
|
+
# A source-backed world may expose fewer raw service endpoints than the agent
|
|
347
|
+
# has model-facing tools because the worker supplies session state and local
|
|
348
|
+
# state-machine operations. The contract-facing specs are what the UI means by
|
|
349
|
+
# tools; counting only handlers made a complete environment look incomplete.
|
|
350
|
+
"tools": len(manifest.get("tool_specs") or manifest.get("tools") or []),
|
|
351
|
+
"sub_goals": held.get("sub_goals", 0),
|
|
352
|
+
"scenarios": held.get("scenarios", 0),
|
|
353
|
+
"runs": len(runs),
|
|
354
|
+
"runs_passed": sum(
|
|
355
|
+
1
|
|
356
|
+
for run in runs
|
|
357
|
+
if run.get("scenarios")
|
|
358
|
+
and run.get("passed") == run.get("scenarios")
|
|
359
|
+
),
|
|
360
|
+
}
|
|
361
|
+
)
|
|
362
|
+
return found
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def _read_json(path: Path) -> dict[str, Any]:
|
|
366
|
+
if not path.exists():
|
|
367
|
+
return {}
|
|
368
|
+
try:
|
|
369
|
+
loaded = json.loads(path.read_text(encoding="utf-8"))
|
|
370
|
+
except ValueError:
|
|
371
|
+
return {}
|
|
372
|
+
return loaded if isinstance(loaded, dict) else {}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""The prompt that drives the simulated person, and filling it in for one scenario.
|
|
2
|
+
|
|
3
|
+
Written once for a conversational agent with its slots left open, so a scenario supplies only
|
|
4
|
+
what differs: who this person is this time and what they are trying to do. What a good one says
|
|
5
|
+
is judgement and lives in the build skill; what is here is only saving it, reading it back, and
|
|
6
|
+
substituting a scenario's values into it.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
SIMULATOR = "simulator_prompt.md"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def save_simulator_prompt(prompt: str, destination: Path) -> Path:
|
|
18
|
+
destination = Path(destination)
|
|
19
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
20
|
+
path = destination / SIMULATOR
|
|
21
|
+
path.write_text(prompt, encoding="utf-8")
|
|
22
|
+
return path
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def load_simulator_prompt(destination: Path) -> str:
|
|
26
|
+
path = Path(destination) / SIMULATOR
|
|
27
|
+
return path.read_text(encoding="utf-8") if path.exists() else ""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def variables_in(prompt: str) -> set[str]:
|
|
31
|
+
"""The slots a scenario has to fill.
|
|
32
|
+
|
|
33
|
+
Written ``{{ name }}``, so the prompt stays readable as prose and a missing value is caught
|
|
34
|
+
before a call is placed rather than appearing verbatim in what the simulated caller says.
|
|
35
|
+
"""
|
|
36
|
+
import re
|
|
37
|
+
|
|
38
|
+
return set(re.findall(r"\{\{\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*\}\}", prompt))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def fill(prompt: str, values: dict[str, Any]) -> tuple[str, list[str]]:
|
|
42
|
+
"""The simulator prompt for one scenario, and anything it left unfilled."""
|
|
43
|
+
import re
|
|
44
|
+
|
|
45
|
+
missing = sorted(variables_in(prompt) - set(values))
|
|
46
|
+
|
|
47
|
+
def swap(match: re.Match[str]) -> str:
|
|
48
|
+
return str(values.get(match.group(1), match.group(0)))
|
|
49
|
+
|
|
50
|
+
filled = re.sub(r"\{\{\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*\}\}", swap, prompt)
|
|
51
|
+
return filled, missing
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def validate_simulator_prompt(
|
|
55
|
+
prompt: str, *, require_persona: bool = False
|
|
56
|
+
) -> list[str]:
|
|
57
|
+
"""Problems that make a simulator prompt unusable.
|
|
58
|
+
|
|
59
|
+
Deliberately thin. What a good simulator prompt says is judgement, and belongs in the skill;
|
|
60
|
+
what can be checked here is that it exists and that a scenario has somewhere to put its
|
|
61
|
+
instruction, since a prompt with no variables is the same prompt for every scenario.
|
|
62
|
+
"""
|
|
63
|
+
problems: list[str] = []
|
|
64
|
+
if len(prompt.strip()) < 80:
|
|
65
|
+
problems.append("too short to be a simulator prompt")
|
|
66
|
+
if not variables_in(prompt):
|
|
67
|
+
problems.append(
|
|
68
|
+
"no variables: without a slot for the scenario's instruction, every scenario would "
|
|
69
|
+
"run the same conversation. Write them as {{ instruction }}"
|
|
70
|
+
)
|
|
71
|
+
if require_persona and "persona" not in variables_in(prompt):
|
|
72
|
+
problems.append(
|
|
73
|
+
"no persona slot: conversational scenarios need {{ persona }} so each caller's "
|
|
74
|
+
"identity and communication profile is explicit"
|
|
75
|
+
)
|
|
76
|
+
return problems
|