agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""JSONL transcript recorder + reader for live lanes (ARCH Decision 3).
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only. The file IS the ledger: one JSON object per line,
|
|
4
|
+
``{"t": <monotonic-relative s>, "channel": ..., "type": ..., "payload": ...}``
|
|
5
|
+
with ``channel`` in {"user", "agent", "tool", "lane"}. Redaction runs at
|
|
6
|
+
write time — declared ``required_env`` VALUES never hit disk. The recorder
|
|
7
|
+
owns the size cap (``AGENT_LEARNING_LIVE_TRANSCRIPT_MAX_BYTES``, default
|
|
8
|
+
64 MiB/scenario): over-cap behavior is retain head+tail, never silently drop.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
from collections import deque
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any, Mapping, Sequence
|
|
20
|
+
|
|
21
|
+
TRANSCRIPT_MAX_BYTES_ENV = "AGENT_LEARNING_LIVE_TRANSCRIPT_MAX_BYTES"
|
|
22
|
+
DEFAULT_TRANSCRIPT_MAX_BYTES = 64 * 1024 * 1024
|
|
23
|
+
TRANSCRIPT_CHANNELS = ("user", "agent", "tool", "lane")
|
|
24
|
+
|
|
25
|
+
# Bytes reserved out of the tail budget for the truncation marker line.
|
|
26
|
+
_TRUNCATION_MARKER_RESERVE = 512
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def redact_env_values(text: str, required_env: Sequence[str]) -> str:
|
|
30
|
+
"""Replace any occurrence of a declared env var's VALUE with
|
|
31
|
+
'[redacted:<NAME>]'. Extends the existing redacted-auth evidence pattern
|
|
32
|
+
(trinity evaluation-hook gates carry auth.redacted evidence) to lane
|
|
33
|
+
transcripts."""
|
|
34
|
+
|
|
35
|
+
for name in required_env:
|
|
36
|
+
value = os.environ.get(name)
|
|
37
|
+
if value:
|
|
38
|
+
text = text.replace(value, f"[redacted:{name}]")
|
|
39
|
+
return text
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def transcript_max_bytes() -> int:
|
|
43
|
+
"""Resolve the per-scenario transcript size cap (3A owns this env var)."""
|
|
44
|
+
|
|
45
|
+
raw = os.environ.get(TRANSCRIPT_MAX_BYTES_ENV)
|
|
46
|
+
if raw:
|
|
47
|
+
try:
|
|
48
|
+
value = int(raw)
|
|
49
|
+
if value > 0:
|
|
50
|
+
return value
|
|
51
|
+
except ValueError:
|
|
52
|
+
pass
|
|
53
|
+
return DEFAULT_TRANSCRIPT_MAX_BYTES
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class TranscriptRecorder:
|
|
57
|
+
"""Append-only JSONL replay transcript (R§3.1, DFAH lineage).
|
|
58
|
+
|
|
59
|
+
One line per event. The dual-channel requirement for voice lanes is just
|
|
60
|
+
two channels of this stream. Capture-to-fixture re-reads the file
|
|
61
|
+
verbatim, so the recorder never writes secrets: every serialized event is
|
|
62
|
+
passed through :func:`redact_env_values` before write.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
def __init__(
|
|
66
|
+
self,
|
|
67
|
+
path: str | Path,
|
|
68
|
+
*,
|
|
69
|
+
required_env: Sequence[str],
|
|
70
|
+
max_bytes: int | None = None,
|
|
71
|
+
) -> None:
|
|
72
|
+
self.path = Path(path)
|
|
73
|
+
self.required_env = tuple(required_env)
|
|
74
|
+
self._max_bytes = int(max_bytes) if max_bytes else transcript_max_bytes()
|
|
75
|
+
self._head_budget = max(self._max_bytes // 2, 1)
|
|
76
|
+
self._tail_budget = max(
|
|
77
|
+
self._max_bytes - self._head_budget - _TRUNCATION_MARKER_RESERVE, 1
|
|
78
|
+
)
|
|
79
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
80
|
+
self._fh = open(self.path, "a", encoding="utf-8")
|
|
81
|
+
self._t0 = time.monotonic()
|
|
82
|
+
self._bytes_written = 0
|
|
83
|
+
self._original_bytes = 0
|
|
84
|
+
self._original_sha = hashlib.sha256()
|
|
85
|
+
self._event_count = 0
|
|
86
|
+
self._channels: set[str] = set()
|
|
87
|
+
self._buffering_tail = False
|
|
88
|
+
self._tail: deque[tuple[int, str]] = deque()
|
|
89
|
+
self._tail_bytes = 0
|
|
90
|
+
self._dropped_events = 0
|
|
91
|
+
self._closed = False
|
|
92
|
+
self._summary: dict[str, Any] | None = None
|
|
93
|
+
# In-memory (already redacted) copies for attribution/stats. Control
|
|
94
|
+
# events are small; bulk audio lives in side files, never in events.
|
|
95
|
+
self.events: list[dict[str, Any]] = []
|
|
96
|
+
|
|
97
|
+
# -- write path ---------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
def record(
|
|
100
|
+
self, channel: str, type: str, payload: Mapping[str, Any]
|
|
101
|
+
) -> None:
|
|
102
|
+
if self._closed:
|
|
103
|
+
raise RuntimeError(f"transcript {self.path} is closed")
|
|
104
|
+
event = {
|
|
105
|
+
"t": round(time.monotonic() - self._t0, 6),
|
|
106
|
+
"channel": str(channel),
|
|
107
|
+
"type": str(type),
|
|
108
|
+
"payload": payload if isinstance(payload, dict) else dict(payload),
|
|
109
|
+
}
|
|
110
|
+
line = json.dumps(event, ensure_ascii=False, default=str)
|
|
111
|
+
line = redact_env_values(line, self.required_env)
|
|
112
|
+
try:
|
|
113
|
+
stored = json.loads(line)
|
|
114
|
+
except ValueError:
|
|
115
|
+
stored = {
|
|
116
|
+
"t": event["t"],
|
|
117
|
+
"channel": event["channel"],
|
|
118
|
+
"type": event["type"],
|
|
119
|
+
"payload": {"unserializable": True},
|
|
120
|
+
}
|
|
121
|
+
self.events.append(stored)
|
|
122
|
+
data = line + "\n"
|
|
123
|
+
nbytes = len(data.encode("utf-8"))
|
|
124
|
+
self._event_count += 1
|
|
125
|
+
self._channels.add(event["channel"])
|
|
126
|
+
self._original_bytes += nbytes
|
|
127
|
+
self._original_sha.update(data.encode("utf-8"))
|
|
128
|
+
if not self._buffering_tail:
|
|
129
|
+
if self._bytes_written + nbytes <= self._head_budget:
|
|
130
|
+
self._fh.write(data)
|
|
131
|
+
self._bytes_written += nbytes
|
|
132
|
+
return
|
|
133
|
+
self._buffering_tail = True
|
|
134
|
+
self._tail.append((nbytes, data))
|
|
135
|
+
self._tail_bytes += nbytes
|
|
136
|
+
while self._tail and self._tail_bytes > self._tail_budget:
|
|
137
|
+
dropped_bytes, _ = self._tail.popleft()
|
|
138
|
+
self._tail_bytes -= dropped_bytes
|
|
139
|
+
self._dropped_events += 1
|
|
140
|
+
|
|
141
|
+
# -- close + summary ----------------------------------------------------
|
|
142
|
+
|
|
143
|
+
def close(self) -> dict[str, Any]:
|
|
144
|
+
"""Flush the tail, close the file, and return the artifact summary:
|
|
145
|
+
``{path, event_count, channels, duration_s, bytes, sha256, complete}``
|
|
146
|
+
plus a ``truncated`` stanza when the cap dropped events."""
|
|
147
|
+
|
|
148
|
+
if self._closed:
|
|
149
|
+
assert self._summary is not None
|
|
150
|
+
return self._summary
|
|
151
|
+
truncated = self._dropped_events > 0
|
|
152
|
+
if self._buffering_tail:
|
|
153
|
+
if truncated:
|
|
154
|
+
marker = json.dumps(
|
|
155
|
+
{
|
|
156
|
+
"t": round(time.monotonic() - self._t0, 6),
|
|
157
|
+
"channel": "lane",
|
|
158
|
+
"type": "transcript_truncated",
|
|
159
|
+
"payload": {
|
|
160
|
+
"retained": "head_and_tail",
|
|
161
|
+
"dropped_events": self._dropped_events,
|
|
162
|
+
},
|
|
163
|
+
},
|
|
164
|
+
ensure_ascii=False,
|
|
165
|
+
)
|
|
166
|
+
self._fh.write(marker + "\n")
|
|
167
|
+
self._bytes_written += len((marker + "\n").encode("utf-8"))
|
|
168
|
+
for nbytes, data in self._tail:
|
|
169
|
+
self._fh.write(data)
|
|
170
|
+
self._bytes_written += nbytes
|
|
171
|
+
self._tail.clear()
|
|
172
|
+
self._tail_bytes = 0
|
|
173
|
+
self._fh.close()
|
|
174
|
+
self._closed = True
|
|
175
|
+
file_sha = hashlib.sha256()
|
|
176
|
+
try:
|
|
177
|
+
with open(self.path, "rb") as handle:
|
|
178
|
+
for chunk in iter(lambda: handle.read(1 << 20), b""):
|
|
179
|
+
file_sha.update(chunk)
|
|
180
|
+
bytes_on_disk = self.path.stat().st_size
|
|
181
|
+
except OSError:
|
|
182
|
+
bytes_on_disk = self._bytes_written
|
|
183
|
+
summary: dict[str, Any] = {
|
|
184
|
+
"path": str(self.path),
|
|
185
|
+
"event_count": self._event_count,
|
|
186
|
+
"channels": sorted(self._channels),
|
|
187
|
+
"duration_s": round(time.monotonic() - self._t0, 6),
|
|
188
|
+
"bytes": bytes_on_disk,
|
|
189
|
+
"sha256": file_sha.hexdigest(),
|
|
190
|
+
"complete": not truncated,
|
|
191
|
+
}
|
|
192
|
+
if truncated:
|
|
193
|
+
summary["truncated"] = {
|
|
194
|
+
"original_bytes": self._original_bytes,
|
|
195
|
+
"original_sha256": self._original_sha.hexdigest(),
|
|
196
|
+
"retained": "head_and_tail",
|
|
197
|
+
"dropped_events": self._dropped_events,
|
|
198
|
+
}
|
|
199
|
+
self._summary = summary
|
|
200
|
+
return summary
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def read_transcript(path: str | Path) -> list[dict[str, Any]]:
|
|
204
|
+
"""Read a JSONL transcript back into its event list (the replay source).
|
|
205
|
+
|
|
206
|
+
Unparseable lines are surfaced as ``transcript_unreadable_line`` lane
|
|
207
|
+
events rather than silently skipped — attribution treats an unreadable
|
|
208
|
+
transcript as lane_infra evidence.
|
|
209
|
+
"""
|
|
210
|
+
|
|
211
|
+
events: list[dict[str, Any]] = []
|
|
212
|
+
with open(path, "r", encoding="utf-8") as handle:
|
|
213
|
+
for line_number, line in enumerate(handle, start=1):
|
|
214
|
+
line = line.strip()
|
|
215
|
+
if not line:
|
|
216
|
+
continue
|
|
217
|
+
try:
|
|
218
|
+
event = json.loads(line)
|
|
219
|
+
except ValueError:
|
|
220
|
+
events.append(
|
|
221
|
+
{
|
|
222
|
+
"t": None,
|
|
223
|
+
"channel": "lane",
|
|
224
|
+
"type": "transcript_unreadable_line",
|
|
225
|
+
"payload": {"line_number": line_number},
|
|
226
|
+
}
|
|
227
|
+
)
|
|
228
|
+
continue
|
|
229
|
+
if isinstance(event, dict):
|
|
230
|
+
events.append(event)
|
|
231
|
+
else:
|
|
232
|
+
events.append(
|
|
233
|
+
{
|
|
234
|
+
"t": None,
|
|
235
|
+
"channel": "lane",
|
|
236
|
+
"type": "transcript_unreadable_line",
|
|
237
|
+
"payload": {"line_number": line_number},
|
|
238
|
+
}
|
|
239
|
+
)
|
|
240
|
+
return events
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Subprocess ENTRY MODULES for live lanes (P3-D1).
|
|
2
|
+
|
|
3
|
+
These files are only ever executed as subprocesses
|
|
4
|
+
(``sys.executable path/to/worker.py``) by ``_runner.spawn_lane_subprocess``
|
|
5
|
+
with the lane extra installed; the release process never imports them. They
|
|
6
|
+
are the ONLY sanctioned home for framework imports under the
|
|
7
|
+
live_lane_boundary gate (workers here keep even those lazy, inside function
|
|
8
|
+
bodies, so the files also import clean without any extra).
|
|
9
|
+
"""
|
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
"""A2A lane worker (3E) — untrusted subprocess entry (P3-D1).
|
|
2
|
+
|
|
3
|
+
Doubles as the loopback A2A peer entry (peer mode). In client mode it walks
|
|
4
|
+
the protocol stages against a peer — card discovery → task lifecycle →
|
|
5
|
+
artifact exchange (R§1 #18) — spawning its own peer-mode sibling on
|
|
6
|
+
127.0.0.1 when no remote peer URL is given (the shipped loopback default
|
|
7
|
+
tier). In peer mode it serves a deterministic echo agent over the REAL A2A
|
|
8
|
+
HTTP protocol.
|
|
9
|
+
|
|
10
|
+
IPC with the harness (client mode): see livekit_worker.py — same
|
|
11
|
+
one-boot-line / JSONL contract. The peer subprocess receives its own boot
|
|
12
|
+
line (``{"type": "boot", "mode": "peer", "port": N}``) from the client.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import asyncio
|
|
18
|
+
import hashlib
|
|
19
|
+
import json
|
|
20
|
+
import os
|
|
21
|
+
import socket
|
|
22
|
+
import subprocess
|
|
23
|
+
import sys
|
|
24
|
+
import time
|
|
25
|
+
import traceback
|
|
26
|
+
import uuid
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _emit(channel: str, type_: str, payload: dict[str, Any]) -> None:
|
|
31
|
+
print(
|
|
32
|
+
json.dumps(
|
|
33
|
+
{"channel": channel, "type": type_, "payload": payload},
|
|
34
|
+
ensure_ascii=False,
|
|
35
|
+
default=str,
|
|
36
|
+
),
|
|
37
|
+
flush=True,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _read_boot() -> dict[str, Any]:
|
|
42
|
+
line = sys.stdin.readline()
|
|
43
|
+
if not line.strip():
|
|
44
|
+
raise RuntimeError("missing boot message on stdin")
|
|
45
|
+
boot = json.loads(line)
|
|
46
|
+
if not isinstance(boot, dict) or boot.get("type") != "boot":
|
|
47
|
+
raise RuntimeError("first stdin line must be a boot message")
|
|
48
|
+
return boot
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _capability_hash(framework: str, version: str) -> str:
|
|
52
|
+
return hashlib.sha256(f"{framework}:{version}".encode("utf-8")).hexdigest()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _free_port() -> int:
|
|
56
|
+
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
|
57
|
+
sock.bind(("127.0.0.1", 0))
|
|
58
|
+
return int(sock.getsockname()[1])
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# --- peer mode ----------------------------------------------------------------
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _run_peer(boot: dict[str, Any]) -> None:
|
|
65
|
+
import uvicorn
|
|
66
|
+
from a2a.server.agent_execution import AgentExecutor
|
|
67
|
+
from a2a.server.apps import A2AStarletteApplication
|
|
68
|
+
from a2a.server.request_handlers import DefaultRequestHandler
|
|
69
|
+
from a2a.server.tasks import InMemoryTaskStore
|
|
70
|
+
from a2a.types import AgentCapabilities, AgentCard, AgentSkill
|
|
71
|
+
from a2a.utils import new_agent_text_message
|
|
72
|
+
|
|
73
|
+
port = int(boot.get("port") or 0) or _free_port()
|
|
74
|
+
|
|
75
|
+
class _EchoExecutor(AgentExecutor):
|
|
76
|
+
"""Deterministic loopback peer behavior: echo the user text."""
|
|
77
|
+
|
|
78
|
+
async def execute(self, context: Any, event_queue: Any) -> None:
|
|
79
|
+
text = ""
|
|
80
|
+
try:
|
|
81
|
+
text = context.get_user_input()
|
|
82
|
+
except Exception:
|
|
83
|
+
pass
|
|
84
|
+
await event_queue.enqueue_event(
|
|
85
|
+
new_agent_text_message(f"echo: {text}")
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
async def cancel(self, context: Any, event_queue: Any) -> None:
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
skill = AgentSkill(
|
|
92
|
+
id="echo",
|
|
93
|
+
name="Echo",
|
|
94
|
+
description="Echoes the inbound message text (deterministic).",
|
|
95
|
+
tags=["echo", "loopback"],
|
|
96
|
+
)
|
|
97
|
+
card = AgentCard(
|
|
98
|
+
name="agent-learning-loopback-peer",
|
|
99
|
+
description="Credential-free loopback A2A peer shipped with the kit.",
|
|
100
|
+
url=f"http://127.0.0.1:{port}/",
|
|
101
|
+
version="1.0.0",
|
|
102
|
+
default_input_modes=["text"],
|
|
103
|
+
default_output_modes=["text"],
|
|
104
|
+
capabilities=AgentCapabilities(streaming=False),
|
|
105
|
+
skills=[skill],
|
|
106
|
+
)
|
|
107
|
+
handler = DefaultRequestHandler(
|
|
108
|
+
agent_executor=_EchoExecutor(), task_store=InMemoryTaskStore()
|
|
109
|
+
)
|
|
110
|
+
application = A2AStarletteApplication(agent_card=card, http_handler=handler)
|
|
111
|
+
uvicorn.run(
|
|
112
|
+
application.build(), host="127.0.0.1", port=port, log_level="error"
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# --- client mode ----------------------------------------------------------------
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _extract_texts(value: Any, into: list[str]) -> None:
|
|
120
|
+
"""Best-effort recursive text extraction across a2a-sdk event shapes."""
|
|
121
|
+
|
|
122
|
+
if value is None:
|
|
123
|
+
return
|
|
124
|
+
if isinstance(value, str):
|
|
125
|
+
if value.strip():
|
|
126
|
+
into.append(value)
|
|
127
|
+
return
|
|
128
|
+
if isinstance(value, (list, tuple)):
|
|
129
|
+
for item in value:
|
|
130
|
+
_extract_texts(item, into)
|
|
131
|
+
return
|
|
132
|
+
for attribute in ("text", "parts", "artifacts", "history", "root", "message", "status"):
|
|
133
|
+
if hasattr(value, attribute):
|
|
134
|
+
_extract_texts(getattr(value, attribute), into)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
async def _run_client(boot: dict[str, Any]) -> None:
|
|
138
|
+
import importlib.metadata
|
|
139
|
+
|
|
140
|
+
import httpx
|
|
141
|
+
|
|
142
|
+
import a2a as a2a_pkg
|
|
143
|
+
from a2a.client import A2ACardResolver
|
|
144
|
+
|
|
145
|
+
version = importlib.metadata.version("a2a-sdk")
|
|
146
|
+
_emit(
|
|
147
|
+
"lane",
|
|
148
|
+
"framework_ready",
|
|
149
|
+
{
|
|
150
|
+
"framework": "a2a-sdk",
|
|
151
|
+
"framework_version": version,
|
|
152
|
+
"capability_hash": _capability_hash("a2a-sdk", version),
|
|
153
|
+
"package_paths": [os.path.dirname(a2a_pkg.__file__)],
|
|
154
|
+
},
|
|
155
|
+
)
|
|
156
|
+
config = boot.get("config") or {}
|
|
157
|
+
stages = [str(stage) for stage in (config.get("stages") or [])]
|
|
158
|
+
message_text = str(config.get("message") or "ping from the harness")
|
|
159
|
+
peer_url = config.get("peer_url")
|
|
160
|
+
peer_process: subprocess.Popen[str] | None = None
|
|
161
|
+
checks: dict[str, bool] = {}
|
|
162
|
+
try:
|
|
163
|
+
if not peer_url:
|
|
164
|
+
port = _free_port()
|
|
165
|
+
peer_process = subprocess.Popen(
|
|
166
|
+
[sys.executable, os.path.abspath(__file__)],
|
|
167
|
+
stdin=subprocess.PIPE,
|
|
168
|
+
stdout=subprocess.DEVNULL,
|
|
169
|
+
stderr=subprocess.DEVNULL,
|
|
170
|
+
text=True,
|
|
171
|
+
env=dict(os.environ),
|
|
172
|
+
)
|
|
173
|
+
assert peer_process.stdin is not None
|
|
174
|
+
peer_process.stdin.write(
|
|
175
|
+
json.dumps({"type": "boot", "mode": "peer", "port": port}) + "\n"
|
|
176
|
+
)
|
|
177
|
+
peer_process.stdin.flush()
|
|
178
|
+
peer_url = f"http://127.0.0.1:{port}"
|
|
179
|
+
|
|
180
|
+
base_url = str(peer_url).rstrip("/")
|
|
181
|
+
async with httpx.AsyncClient(timeout=30.0) as http_client:
|
|
182
|
+
# Wait for the peer to come up (loopback boot is fast but async).
|
|
183
|
+
card_paths = (
|
|
184
|
+
"/.well-known/agent-card.json",
|
|
185
|
+
"/.well-known/agent.json",
|
|
186
|
+
)
|
|
187
|
+
deadline = time.monotonic() + 30.0
|
|
188
|
+
reachable = False
|
|
189
|
+
while time.monotonic() < deadline and not reachable:
|
|
190
|
+
for path in card_paths:
|
|
191
|
+
try:
|
|
192
|
+
response = await http_client.get(base_url + path)
|
|
193
|
+
if response.status_code == 200:
|
|
194
|
+
reachable = True
|
|
195
|
+
break
|
|
196
|
+
except httpx.HTTPError:
|
|
197
|
+
pass
|
|
198
|
+
if not reachable:
|
|
199
|
+
await asyncio.sleep(0.2)
|
|
200
|
+
if not reachable:
|
|
201
|
+
raise RuntimeError(f"A2A peer at {base_url} never became reachable")
|
|
202
|
+
|
|
203
|
+
# --- stage: card discovery -----------------------------------
|
|
204
|
+
resolver = A2ACardResolver(http_client, base_url)
|
|
205
|
+
card = await resolver.get_agent_card()
|
|
206
|
+
card_ok = bool(getattr(card, "name", None))
|
|
207
|
+
_emit(
|
|
208
|
+
"agent",
|
|
209
|
+
"protocol_stage",
|
|
210
|
+
{"stage": "card_discovery", "ok": card_ok, "peer": getattr(card, "name", None)},
|
|
211
|
+
)
|
|
212
|
+
if "card_discovery" in stages:
|
|
213
|
+
checks["card_discovery"] = card_ok
|
|
214
|
+
|
|
215
|
+
# --- stage: task lifecycle + artifact exchange ----------------
|
|
216
|
+
texts: list[str] = []
|
|
217
|
+
lifecycle_ok = False
|
|
218
|
+
try:
|
|
219
|
+
from a2a.client import ClientConfig, ClientFactory
|
|
220
|
+
from a2a.types import Message, Part, Role, TextPart
|
|
221
|
+
|
|
222
|
+
factory = ClientFactory(ClientConfig(httpx_client=http_client))
|
|
223
|
+
client = factory.create(card)
|
|
224
|
+
try:
|
|
225
|
+
outbound = Message(
|
|
226
|
+
role=Role.user,
|
|
227
|
+
parts=[Part(root=TextPart(text=message_text))],
|
|
228
|
+
message_id=uuid.uuid4().hex,
|
|
229
|
+
)
|
|
230
|
+
except TypeError:
|
|
231
|
+
outbound = Message(
|
|
232
|
+
role=Role.user,
|
|
233
|
+
parts=[Part(root=TextPart(text=message_text))],
|
|
234
|
+
messageId=uuid.uuid4().hex,
|
|
235
|
+
)
|
|
236
|
+
_emit("user", "message", {"text": message_text})
|
|
237
|
+
async for event in client.send_message(outbound):
|
|
238
|
+
lifecycle_ok = True
|
|
239
|
+
_extract_texts(event, texts)
|
|
240
|
+
except ImportError:
|
|
241
|
+
# Older SDK line: single-shot A2AClient JSON-RPC surface.
|
|
242
|
+
from a2a.client import A2AClient
|
|
243
|
+
from a2a.types import (
|
|
244
|
+
Message,
|
|
245
|
+
MessageSendParams,
|
|
246
|
+
Part,
|
|
247
|
+
Role,
|
|
248
|
+
SendMessageRequest,
|
|
249
|
+
TextPart,
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
client = A2AClient(httpx_client=http_client, agent_card=card)
|
|
253
|
+
request = SendMessageRequest(
|
|
254
|
+
id=uuid.uuid4().hex,
|
|
255
|
+
params=MessageSendParams(
|
|
256
|
+
message=Message(
|
|
257
|
+
role=Role.user,
|
|
258
|
+
parts=[Part(root=TextPart(text=message_text))],
|
|
259
|
+
messageId=uuid.uuid4().hex,
|
|
260
|
+
)
|
|
261
|
+
),
|
|
262
|
+
)
|
|
263
|
+
_emit("user", "message", {"text": message_text})
|
|
264
|
+
response = await client.send_message(request)
|
|
265
|
+
lifecycle_ok = response is not None
|
|
266
|
+
_extract_texts(response, texts)
|
|
267
|
+
|
|
268
|
+
reply = next((text for text in texts if "echo" in text.lower()), "")
|
|
269
|
+
artifact_ok = bool(reply) or any(text.strip() for text in texts)
|
|
270
|
+
_emit(
|
|
271
|
+
"agent",
|
|
272
|
+
"protocol_stage",
|
|
273
|
+
{"stage": "task_lifecycle", "ok": lifecycle_ok},
|
|
274
|
+
)
|
|
275
|
+
_emit(
|
|
276
|
+
"agent",
|
|
277
|
+
"protocol_stage",
|
|
278
|
+
{
|
|
279
|
+
"stage": "artifact_exchange",
|
|
280
|
+
"ok": artifact_ok,
|
|
281
|
+
"text": (reply or " ".join(texts))[:500],
|
|
282
|
+
},
|
|
283
|
+
)
|
|
284
|
+
if "task_lifecycle" in stages:
|
|
285
|
+
checks["task_lifecycle"] = lifecycle_ok
|
|
286
|
+
if "artifact_exchange" in stages:
|
|
287
|
+
checks["artifact_exchange"] = artifact_ok
|
|
288
|
+
|
|
289
|
+
passed = bool(checks) and all(checks.values())
|
|
290
|
+
_emit("lane", "verification", {"passed": passed, "checks": checks})
|
|
291
|
+
finally:
|
|
292
|
+
if peer_process is not None:
|
|
293
|
+
peer_process.terminate()
|
|
294
|
+
try:
|
|
295
|
+
peer_process.wait(timeout=10)
|
|
296
|
+
except subprocess.TimeoutExpired:
|
|
297
|
+
peer_process.kill()
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def main() -> int:
|
|
301
|
+
boot = _read_boot()
|
|
302
|
+
mode = str(boot.get("mode") or "client")
|
|
303
|
+
try:
|
|
304
|
+
if mode == "peer":
|
|
305
|
+
_run_peer(boot)
|
|
306
|
+
else:
|
|
307
|
+
asyncio.run(_run_client(boot))
|
|
308
|
+
except Exception:
|
|
309
|
+
_emit("lane", "worker_error", {"traceback": traceback.format_exc()})
|
|
310
|
+
traceback.print_exc(file=sys.stderr)
|
|
311
|
+
return 1
|
|
312
|
+
return 0
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
if __name__ == "__main__":
|
|
316
|
+
sys.exit(main())
|