dd-trace 6.16.0 → 6.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ci/diagnose.js +0 -23
- package/ci/test-optimization-validation/approval-artifacts.js +0 -3
- package/ci/test-optimization-validation/approval.js +0 -6
- package/ci/test-optimization-validation/artifact-id.js +0 -1
- package/ci/test-optimization-validation/bounded-json.js +0 -1
- package/ci/test-optimization-validation/cli.js +0 -12
- package/ci/test-optimization-validation/command-output-policy.js +0 -1
- package/ci/test-optimization-validation/command-runner.js +0 -3
- package/ci/test-optimization-validation/environment.js +0 -4
- package/ci/test-optimization-validation/executable-approval.js +0 -1
- package/ci/test-optimization-validation/executable.js +0 -5
- package/ci/test-optimization-validation/framework-adapters/cucumber.js +0 -6
- package/ci/test-optimization-validation/framework-adapters/cypress.js +0 -3
- package/ci/test-optimization-validation/framework-adapters/playwright.js +0 -6
- package/ci/test-optimization-validation/framework-adapters/vitest.js +0 -4
- package/ci/test-optimization-validation/generated-test-contract.js +0 -5
- package/ci/test-optimization-validation/generated-verifier.js +0 -1
- package/ci/test-optimization-validation/literal-environment.js +0 -1
- package/ci/test-optimization-validation/manifest-scaffold.js +0 -17
- package/ci/test-optimization-validation/manifest-schema.js +0 -21
- package/ci/test-optimization-validation/offline-fixtures.js +0 -4
- package/ci/test-optimization-validation/plan-writer.js +0 -11
- package/ci/test-optimization-validation/preflight-runner.js +0 -1
- package/ci/test-optimization-validation/redaction.js +0 -4
- package/ci/test-optimization-validation/report-writer.js +0 -16
- package/ci/test-optimization-validation/runner-command.js +0 -3
- package/ci/test-optimization-validation/runner-contract.js +0 -4
- package/ci/test-optimization-validation/safe-files.js +0 -1
- package/ci/test-optimization-validation/scenarios/basic-reporting.js +0 -1
- package/ci/test-optimization-validation/scenarios/ci-wiring.js +0 -1
- package/ci/test-optimization-validation/source-text.js +0 -3
- package/ci/vitest-no-worker-init-setup.mjs +41 -16
- package/index.d.ts +208 -0
- package/package.json +17 -16
- package/packages/datadog-esbuild/src/utils.js +0 -2
- package/packages/datadog-instrumentations/src/aerospike.js +28 -6
- package/packages/datadog-instrumentations/src/aws-durable-execution-sdk-js.js +0 -2
- package/packages/datadog-instrumentations/src/aws-sdk.js +0 -2
- package/packages/datadog-instrumentations/src/claude-agent-sdk.js +0 -2
- package/packages/datadog-instrumentations/src/console.js +97 -0
- package/packages/datadog-instrumentations/src/cucumber.js +77 -12
- package/packages/datadog-instrumentations/src/cypress-config.js +27 -38
- package/packages/datadog-instrumentations/src/helpers/bundler-register.js +6 -0
- package/packages/datadog-instrumentations/src/helpers/channel.js +0 -2
- package/packages/datadog-instrumentations/src/helpers/check-require-cache.js +2 -1
- package/packages/datadog-instrumentations/src/helpers/graphql-jit-runtime.js +0 -5
- package/packages/datadog-instrumentations/src/helpers/hooks.js +8 -6
- package/packages/datadog-instrumentations/src/helpers/instrumentation-utils.js +0 -3
- package/packages/datadog-instrumentations/src/helpers/pool-acquire.js +0 -6
- package/packages/datadog-instrumentations/src/helpers/register.js +24 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/index.js +111 -26
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentation-registry.js +41 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentations/index.js +3 -17
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentations/playwright.js +55 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentations/postgres.js +36 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentations/supabase.js +128 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/instrumentations/webdriverio.js +13 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/targets.js +28 -3
- package/packages/datadog-instrumentations/src/helpers/rewriter/targets.json +20 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/transforms/postgres.js +388 -0
- package/packages/datadog-instrumentations/src/helpers/rewriter/transforms.js +0 -5
- package/packages/datadog-instrumentations/src/helpers/router-helper.js +0 -1
- package/packages/datadog-instrumentations/src/jest.js +165 -51
- package/packages/datadog-instrumentations/src/knex.js +0 -1
- package/packages/datadog-instrumentations/src/mariadb-bundle.js +0 -24
- package/packages/datadog-instrumentations/src/mariadb.js +0 -2
- package/packages/datadog-instrumentations/src/mocha/common.js +0 -1
- package/packages/datadog-instrumentations/src/mocha/main.js +25 -26
- package/packages/datadog-instrumentations/src/mocha/utils.js +119 -49
- package/packages/datadog-instrumentations/src/mocha/webdriverio-protocol.js +0 -6
- package/packages/datadog-instrumentations/src/mocha/worker.js +14 -17
- package/packages/datadog-instrumentations/src/mysql2.js +0 -2
- package/packages/datadog-instrumentations/src/openai-realtime/audio-accumulator.js +241 -0
- package/packages/datadog-instrumentations/src/openai-realtime/events.js +76 -0
- package/packages/datadog-instrumentations/src/openai-realtime/index.js +301 -0
- package/packages/datadog-instrumentations/src/openai-realtime/session.js +998 -0
- package/packages/datadog-instrumentations/src/openai-realtime/turns.js +211 -0
- package/packages/datadog-instrumentations/src/openai.js +51 -12
- package/packages/datadog-instrumentations/src/oracledb.js +9 -3
- package/packages/datadog-instrumentations/src/playwright-reporter.js +6 -20
- package/packages/datadog-instrumentations/src/playwright.js +280 -56
- package/packages/datadog-instrumentations/src/{azure-cosmos.js → postgres.js} +1 -1
- package/packages/datadog-instrumentations/src/router.js +0 -2
- package/packages/datadog-instrumentations/src/rum-browser-scripts.js +0 -2
- package/packages/datadog-instrumentations/src/sequelize.js +0 -1
- package/packages/datadog-instrumentations/src/supabase.js +15 -0
- package/packages/datadog-instrumentations/src/undici.js +7 -3
- package/packages/datadog-instrumentations/src/vitest-main-no-worker-init.js +18 -18
- package/packages/datadog-instrumentations/src/vitest-main.js +121 -30
- package/packages/datadog-instrumentations/src/vitest-util.js +7 -1
- package/packages/datadog-instrumentations/src/vitest-worker.js +49 -13
- package/packages/datadog-instrumentations/src/webdriverio.js +44 -47
- package/packages/datadog-instrumentations/src/ws.js +41 -2
- package/packages/datadog-plugin-ai/src/utils.js +0 -2
- package/packages/datadog-plugin-aws-durable-execution-sdk-js/src/trace-checkpoint.js +0 -1
- package/packages/datadog-plugin-aws-durable-execution-sdk-js/src/util.js +0 -2
- package/packages/datadog-plugin-aws-sdk/src/base.js +0 -1
- package/packages/datadog-plugin-aws-sdk/src/services/bedrockruntime/utils.js +0 -1
- package/packages/datadog-plugin-aws-sdk/src/services/eventbridge.js +0 -1
- package/packages/datadog-plugin-aws-sdk/src/util.js +0 -1
- package/packages/datadog-plugin-azure-durable-functions/src/index.js +0 -1
- package/packages/datadog-plugin-cucumber/src/index.js +9 -0
- package/packages/datadog-plugin-cypress/src/cypress-plugin.js +211 -85
- package/packages/datadog-plugin-cypress/src/finalization.js +0 -3
- package/packages/datadog-plugin-cypress/src/index.js +3 -1
- package/packages/datadog-plugin-cypress/src/source-map-utils.js +0 -4
- package/packages/datadog-plugin-cypress/src/support.js +52 -7
- package/packages/datadog-plugin-graphql/src/resolve-error.js +1 -2
- package/packages/datadog-plugin-graphql/src/utils.js +0 -5
- package/packages/datadog-plugin-jest/src/index.js +12 -0
- package/packages/datadog-plugin-kafkajs/src/producer.js +0 -1
- package/packages/datadog-plugin-mocha/src/index.js +38 -11
- package/packages/datadog-plugin-mongodb-core/src/query.js +0 -1
- package/packages/datadog-plugin-openai/src/index.js +16 -1
- package/packages/datadog-plugin-openai/src/realtime.js +144 -0
- package/packages/datadog-plugin-openai-agents/src/integration.js +0 -4
- package/packages/datadog-plugin-openai-agents/src/util.js +0 -1
- package/packages/datadog-plugin-playwright/src/index.js +81 -9
- package/packages/datadog-plugin-postgres/src/index.js +167 -0
- package/packages/datadog-plugin-prisma/src/datadog-tracing-helper.js +0 -1
- package/packages/datadog-plugin-redis/src/index.js +15 -1
- package/packages/datadog-plugin-supabase/src/error.js +18 -0
- package/packages/datadog-plugin-supabase/src/index.js +21 -0
- package/packages/datadog-plugin-supabase/src/supabase-auth-js-gotrueclient-getuser.js +69 -0
- package/packages/datadog-plugin-supabase/src/supabase-functions-js-functionsclient-invoke.js +70 -0
- package/packages/datadog-plugin-supabase/src/supabase-postgrest-js-postgrestbuilder-then.js +122 -0
- package/packages/datadog-plugin-supabase/src/supabase-realtime-js-realtimechannel-send.js +66 -0
- package/packages/datadog-plugin-supabase/src/supabase-storage-js-handle-request.js +93 -0
- package/packages/datadog-plugin-supabase/src/url.js +17 -0
- package/packages/datadog-plugin-vitest/src/index.js +9 -0
- package/packages/datadog-plugin-ws/src/producer.js +1 -1
- package/packages/datadog-plugin-ws/src/util.js +0 -4
- package/packages/datadog-shimmer/src/shimmer.js +14 -10
- package/packages/datadog-turbopack/index.js +4 -7
- package/packages/datadog-turbopack/src/loader.js +6 -18
- package/packages/datadog-webpack/src/loader.js +0 -1
- package/packages/dd-trace/src/agent/info.js +5 -4
- package/packages/dd-trace/src/aiguard/client.js +2 -3
- package/packages/dd-trace/src/aiguard/evaluation.js +6 -12
- package/packages/dd-trace/src/aiguard/index.js +2 -2
- package/packages/dd-trace/src/aiguard/integrations/index.js +1 -1
- package/packages/dd-trace/src/aiguard/integrations/openai.js +81 -7
- package/packages/dd-trace/src/aiguard/messages/anthropic.js +0 -2
- package/packages/dd-trace/src/aiguard/messages/openai.js +64 -1
- package/packages/dd-trace/src/aiguard/messages/utils.js +0 -1
- package/packages/dd-trace/src/aiguard/redaction.js +0 -1
- package/packages/dd-trace/src/aiguard/sdk.js +1 -1
- package/packages/dd-trace/src/appsec/activation.js +1 -1
- package/packages/dd-trace/src/appsec/api_security/normalized-route.js +0 -5
- package/packages/dd-trace/src/appsec/api_security/sampler.js +0 -4
- package/packages/dd-trace/src/appsec/blocking/index.js +3 -3
- package/packages/dd-trace/src/appsec/downstream_requests.js +1 -4
- package/packages/dd-trace/src/appsec/iast/index.js +1 -1
- package/packages/dd-trace/src/appsec/iast/overhead-controller.js +5 -5
- package/packages/dd-trace/src/appsec/iast/taint-tracking/index.js +1 -1
- package/packages/dd-trace/src/appsec/iast/taint-tracking/plugin.js +1 -1
- package/packages/dd-trace/src/appsec/iast/taint-tracking/rewriter.js +3 -3
- package/packages/dd-trace/src/appsec/iast/vulnerabilities-formatter/evidence-redaction/sensitive-analyzers/sql-sensitive-analyzer.js +8 -10
- package/packages/dd-trace/src/appsec/iast/vulnerabilities-formatter/evidence-redaction/sensitive-handler.js +2 -2
- package/packages/dd-trace/src/appsec/iast/vulnerabilities-formatter/utils.js +4 -1
- package/packages/dd-trace/src/appsec/iast/vulnerability-reporter.js +7 -7
- package/packages/dd-trace/src/appsec/index.js +2 -2
- package/packages/dd-trace/src/appsec/lambda.js +71 -2
- package/packages/dd-trace/src/appsec/rasp/utils.js +5 -1
- package/packages/dd-trace/src/appsec/remote_config.js +8 -6
- package/packages/dd-trace/src/appsec/reporter.js +10 -6
- package/packages/dd-trace/src/appsec/rule_manager.js +3 -3
- package/packages/dd-trace/src/appsec/sdk/track_event.js +0 -2
- package/packages/dd-trace/src/appsec/waf/index.js +2 -2
- package/packages/dd-trace/src/appsec/waf/waf_manager.js +8 -5
- package/packages/dd-trace/src/carrier.js +0 -5
- package/packages/dd-trace/src/ci-visibility/dynamic-atr-retries.js +104 -0
- package/packages/dd-trace/src/ci-visibility/dynamic-instrumentation/worker/index.js +0 -2
- package/packages/dd-trace/src/ci-visibility/efd-retry-policy.js +34 -3
- package/packages/dd-trace/src/ci-visibility/exporters/agentless/coverage-writer.js +0 -1
- package/packages/dd-trace/src/ci-visibility/exporters/agentless/di-logs-writer.js +0 -1
- package/packages/dd-trace/src/ci-visibility/exporters/agentless/writer.js +0 -1
- package/packages/dd-trace/src/ci-visibility/exporters/ci-validation/index.js +0 -4
- package/packages/dd-trace/src/ci-visibility/exporters/ci-validation/msgpack-to-json.js +0 -1
- package/packages/dd-trace/src/ci-visibility/exporters/ci-validation/sink.js +0 -2
- package/packages/dd-trace/src/ci-visibility/exporters/ci-visibility-exporter.js +37 -15
- package/packages/dd-trace/src/ci-visibility/exporters/git/git_metadata.js +0 -3
- package/packages/dd-trace/src/ci-visibility/exporters/request.js +0 -11
- package/packages/dd-trace/src/ci-visibility/exporters/settings-cache-key.js +0 -1
- package/packages/dd-trace/src/ci-visibility/intelligent-test-runner/get-skippable-suites.js +0 -1
- package/packages/dd-trace/src/ci-visibility/requests/fs-cache.js +0 -6
- package/packages/dd-trace/src/ci-visibility/requests/get-library-configuration.js +2 -2
- package/packages/dd-trace/src/ci-visibility/requests/rate-limit.js +0 -1
- package/packages/dd-trace/src/ci-visibility/requests/upload-test-screenshot.js +0 -5
- package/packages/dd-trace/src/ci-visibility/requests/video-request.js +0 -1
- package/packages/dd-trace/src/ci-visibility/telemetry.js +13 -0
- package/packages/dd-trace/src/ci-visibility/test-optimization-cache.js +0 -1
- package/packages/dd-trace/src/ci-visibility/test-screenshot.js +0 -2
- package/packages/dd-trace/src/ci-visibility/test-video.js +0 -1
- package/packages/dd-trace/src/config/defaults.js +3 -2
- package/packages/dd-trace/src/config/generated-config-types.d.ts +88 -82
- package/packages/dd-trace/src/config/helper.js +0 -1
- package/packages/dd-trace/src/config/index.js +31 -30
- package/packages/dd-trace/src/config/parsers.js +34 -6
- package/packages/dd-trace/src/config/remote_config.js +25 -109
- package/packages/dd-trace/src/config/supported-configurations.json +150 -10
- package/packages/dd-trace/src/datastreams/encoding.js +0 -1
- package/packages/dd-trace/src/datastreams/pathway.js +0 -1
- package/packages/dd-trace/src/debugger/constants.js +10 -0
- package/packages/dd-trace/src/debugger/devtools_client/breakpoints.js +62 -10
- package/packages/dd-trace/src/debugger/devtools_client/config.js +4 -1
- package/packages/dd-trace/src/debugger/devtools_client/guardrail-metrics.js +15 -0
- package/packages/dd-trace/src/debugger/devtools_client/index.js +103 -27
- package/packages/dd-trace/src/debugger/devtools_client/json-buffer.js +50 -7
- package/packages/dd-trace/src/debugger/devtools_client/log.js +3 -1
- package/packages/dd-trace/src/debugger/devtools_client/probe_sampler.js +21 -8
- package/packages/dd-trace/src/debugger/devtools_client/send.js +41 -15
- package/packages/dd-trace/src/debugger/devtools_client/snapshot/index.js +53 -13
- package/packages/dd-trace/src/debugger/devtools_client/snapshot/processor.js +118 -71
- package/packages/dd-trace/src/debugger/devtools_client/snapshot/redaction.js +5 -3
- package/packages/dd-trace/src/debugger/devtools_client/snapshot-pruner.js +0 -3
- package/packages/dd-trace/src/debugger/devtools_client/status.js +21 -7
- package/packages/dd-trace/src/debugger/guardrail-metrics.js +167 -0
- package/packages/dd-trace/src/debugger/index.js +51 -3
- package/packages/dd-trace/src/debugger/inspect-segment.js +0 -3
- package/packages/dd-trace/src/debugger/probe_sampler.js +133 -9
- package/packages/dd-trace/src/debugger/probe_sampler_constants.js +31 -1
- package/packages/dd-trace/src/dogstatsd.js +0 -23
- package/packages/dd-trace/src/encode/0.4-cross-payload.js +0 -3
- package/packages/dd-trace/src/encode/0.4.js +0 -7
- package/packages/dd-trace/src/encode/0.5.js +0 -2
- package/packages/dd-trace/src/encode/tags-processors.js +0 -1
- package/packages/dd-trace/src/evp_proxy/constants.js +2 -0
- package/packages/dd-trace/src/evp_proxy/direct.js +17 -13
- package/packages/dd-trace/src/evp_proxy/discovery.js +14 -11
- package/packages/dd-trace/src/evp_proxy/path.js +21 -9
- package/packages/dd-trace/src/exporters/agent/writer.js +0 -1
- package/packages/dd-trace/src/exporters/agentless/index.js +0 -1
- package/packages/dd-trace/src/exporters/agentless/intake.js +0 -1
- package/packages/dd-trace/src/exporters/agentless/writer.js +0 -3
- package/packages/dd-trace/src/exporters/common/final-flush-request-tracker.js +0 -6
- package/packages/dd-trace/src/exporters/common/request.js +1 -0
- package/packages/dd-trace/src/exporters/common/url.js +33 -4
- package/packages/dd-trace/src/exporters/common/writer.js +0 -2
- package/packages/dd-trace/src/id.js +0 -11
- package/packages/dd-trace/src/llmobs/audio-codec.js +192 -0
- package/packages/dd-trace/src/llmobs/audio-utils.js +121 -0
- package/packages/dd-trace/src/llmobs/constants/audio.js +35 -0
- package/packages/dd-trace/src/llmobs/constants/prompts.js +14 -0
- package/packages/dd-trace/src/llmobs/constants/tags.js +17 -0
- package/packages/dd-trace/src/llmobs/eval-metric.js +0 -6
- package/packages/dd-trace/src/llmobs/experiments/dataset.js +0 -4
- package/packages/dd-trace/src/llmobs/experiments/experiment.js +0 -3
- package/packages/dd-trace/src/llmobs/experiments/index.js +7 -6
- package/packages/dd-trace/src/llmobs/experiments/util.js +1 -11
- package/packages/dd-trace/src/llmobs/index.js +1 -3
- package/packages/dd-trace/src/llmobs/noop.js +10 -0
- package/packages/dd-trace/src/llmobs/plugins/ai/util.js +0 -4
- package/packages/dd-trace/src/llmobs/plugins/anthropic/util.js +0 -1
- package/packages/dd-trace/src/llmobs/plugins/genai/util.js +0 -4
- package/packages/dd-trace/src/llmobs/plugins/modelcontextprotocol-sdk/utils.js +0 -2
- package/packages/dd-trace/src/llmobs/plugins/openai/constants.js +0 -2
- package/packages/dd-trace/src/llmobs/plugins/openai/index.js +4 -6
- package/packages/dd-trace/src/llmobs/plugins/openai/realtime.js +285 -0
- package/packages/dd-trace/src/llmobs/plugins/openai/utils.js +16 -3
- package/packages/dd-trace/src/llmobs/prompts/cache.js +272 -0
- package/packages/dd-trace/src/llmobs/prompts/manager.js +647 -0
- package/packages/dd-trace/src/llmobs/prompts/noop.js +29 -0
- package/packages/dd-trace/src/llmobs/prompts/prompt.js +158 -0
- package/packages/dd-trace/src/llmobs/sdk.js +21 -6
- package/packages/dd-trace/src/llmobs/span_processor.js +81 -2
- package/packages/dd-trace/src/llmobs/tagger.js +39 -11
- package/packages/dd-trace/src/llmobs/telemetry.js +36 -4
- package/packages/dd-trace/src/llmobs/util.js +0 -34
- package/packages/dd-trace/src/llmobs/writers/base.js +0 -1
- package/packages/dd-trace/src/llmobs/writers/util.js +1 -2
- package/packages/dd-trace/src/log/channels.js +3 -0
- package/packages/dd-trace/src/log/index.js +23 -5
- package/packages/dd-trace/src/log-submission/log-submission-plugin.js +74 -6
- package/packages/dd-trace/src/msgpack/chunk.js +0 -1
- package/packages/dd-trace/src/openfeature/agentless_configuration_source.js +0 -12
- package/packages/dd-trace/src/openfeature/configuration_source.js +4 -1
- package/packages/dd-trace/src/openfeature/encoding.js +0 -2
- package/packages/dd-trace/src/openfeature/eval-metrics-hook.js +0 -1
- package/packages/dd-trace/src/openfeature/flagging_provider.js +3 -3
- package/packages/dd-trace/src/openfeature/index.js +6 -6
- package/packages/dd-trace/src/openfeature/span-enrichment-hook.js +0 -1
- package/packages/dd-trace/src/openfeature/span-enrichment.js +0 -4
- package/packages/dd-trace/src/openfeature/writers/base.js +90 -23
- package/packages/dd-trace/src/openfeature/writers/exposures.js +4 -5
- package/packages/dd-trace/src/openfeature/writers/util.js +104 -37
- package/packages/dd-trace/src/opentelemetry/logs/logger.js +0 -1
- package/packages/dd-trace/src/opentelemetry/logs/logger_provider.js +0 -1
- package/packages/dd-trace/src/opentelemetry/logs/otlp_http_log_exporter.js +0 -1
- package/packages/dd-trace/src/opentelemetry/logs/otlp_transformer.js +0 -2
- package/packages/dd-trace/src/opentelemetry/metrics/meter_provider.js +0 -1
- package/packages/dd-trace/src/opentelemetry/metrics/otlp_span_stats_transformer.js +0 -1
- package/packages/dd-trace/src/opentelemetry/metrics/otlp_transformer.js +0 -1
- package/packages/dd-trace/src/opentelemetry/metrics/periodic_metric_reader.js +1 -14
- package/packages/dd-trace/src/opentelemetry/metrics/time.js +0 -1
- package/packages/dd-trace/src/opentelemetry/otlp/otlp_http_exporter_base.js +30 -13
- package/packages/dd-trace/src/opentelemetry/otlp/otlp_transformer_base.js +0 -1
- package/packages/dd-trace/src/opentelemetry/span-helpers.js +0 -1
- package/packages/dd-trace/src/opentelemetry/trace/otlp_http_trace_exporter.js +0 -1
- package/packages/dd-trace/src/opentelemetry/trace/otlp_transformer.js +0 -3
- package/packages/dd-trace/src/opentracing/propagation/text_map.js +0 -6
- package/packages/dd-trace/src/opentracing/span_context.js +0 -1
- package/packages/dd-trace/src/opentracing/tracer.js +3 -3
- package/packages/dd-trace/src/otel-thread-ctx.js +38 -11
- package/packages/dd-trace/src/payload-tagging/tagging.js +15 -3
- package/packages/dd-trace/src/plugin_manager.js +27 -8
- package/packages/dd-trace/src/plugins/ci_plugin.js +1 -2
- package/packages/dd-trace/src/plugins/database.js +2 -6
- package/packages/dd-trace/src/plugins/index.js +7 -0
- package/packages/dd-trace/src/plugins/plugin.js +0 -4
- package/packages/dd-trace/src/plugins/tracing.js +1 -1
- package/packages/dd-trace/src/plugins/util/git.js +1 -3
- package/packages/dd-trace/src/plugins/util/http-otel-semantics.js +0 -1
- package/packages/dd-trace/src/plugins/util/llm.js +0 -2
- package/packages/dd-trace/src/plugins/util/status-validator.js +0 -1
- package/packages/dd-trace/src/plugins/util/test.js +26 -20
- package/packages/dd-trace/src/plugins/util/url.js +0 -5
- package/packages/dd-trace/src/priority_sampler.js +0 -6
- package/packages/dd-trace/src/process-tags/index.js +0 -1
- package/packages/dd-trace/src/profiler.js +132 -35
- package/packages/dd-trace/src/profiling/exporters/event_serializer.js +0 -1
- package/packages/dd-trace/src/profiling/profiler.js +101 -41
- package/packages/dd-trace/src/profiling/ssi-heuristics.js +29 -15
- package/packages/dd-trace/src/propagation-hash/index.js +0 -1
- package/packages/dd-trace/src/proxy.js +23 -47
- package/packages/dd-trace/src/random_sampler.js +0 -4
- package/packages/dd-trace/src/rate_limiter.js +0 -4
- package/packages/dd-trace/src/remote_config/capabilities.js +1 -9
- package/packages/dd-trace/src/remote_config/index.js +0 -1
- package/packages/dd-trace/src/runtime_metrics/otlp_runtime_metrics.js +0 -3
- package/packages/dd-trace/src/sampler.js +0 -4
- package/packages/dd-trace/src/sampling_rule.js +0 -7
- package/packages/dd-trace/src/serverless/vercel.js +0 -1
- package/packages/dd-trace/src/serverless.js +0 -1
- package/packages/dd-trace/src/service-naming/index.js +0 -1
- package/packages/dd-trace/src/service-naming/schemas/util.js +0 -1
- package/packages/dd-trace/src/service-naming/schemas/v0/storage.js +7 -0
- package/packages/dd-trace/src/service-naming/schemas/v1/storage.js +5 -0
- package/packages/dd-trace/src/span_processor.js +2 -2
- package/packages/dd-trace/src/span_sampler.js +0 -1
- package/packages/dd-trace/src/standalone/product.js +2 -2
- package/packages/dd-trace/src/standalone/tracesource.js +0 -1
- package/packages/dd-trace/src/startup-log.js +1 -1
- package/packages/dd-trace/src/telemetry/endpoints.js +0 -8
- package/packages/dd-trace/src/telemetry/logs/index.js +16 -5
- package/packages/dd-trace/src/telemetry/logs/log-collector.js +16 -6
- package/packages/dd-trace/src/telemetry/metrics.js +0 -2
- package/packages/dd-trace/src/telemetry/send-data.js +2 -1
- package/packages/dd-trace/src/telemetry/telemetry.js +4 -1
- package/packages/dd-trace/src/tracer.js +5 -4
- package/packages/dd-trace/src/util.js +13 -1
- package/vendor/dist/@datadog/openfeature-node-server/index.js +1 -1
- package/packages/datadog-instrumentations/src/bullmq.js +0 -11
- package/packages/datadog-instrumentations/src/langchain.js +0 -7
- package/packages/datadog-instrumentations/src/langgraph.js +0 -7
- package/packages/datadog-instrumentations/src/mercurius.js +0 -11
|
@@ -0,0 +1,998 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
|
|
3
|
+
const { randomUUID } = require('node:crypto')
|
|
4
|
+
|
|
5
|
+
const {
|
|
6
|
+
bytesPerSecond,
|
|
7
|
+
realtimeAudioFormatToMime,
|
|
8
|
+
segmentDurationMs,
|
|
9
|
+
} = require('../../../dd-trace/src/llmobs/audio-codec')
|
|
10
|
+
const log = require('../../../dd-trace/src/log')
|
|
11
|
+
const { extractResponseTools, normalizeResponseEventType } = require('./events')
|
|
12
|
+
const { InputTurn, ResponseTurn } = require('./turns')
|
|
13
|
+
|
|
14
|
+
// Realtime PCM is 24 kHz mono by spec; overridden from the session's format object when present.
|
|
15
|
+
const DEFAULT_AUDIO_RATE = 24_000
|
|
16
|
+
|
|
17
|
+
// Ceiling on how long a finished turn is held waiting for its audio to finish playing. A connection
|
|
18
|
+
// that goes idle right after a response is the gap: nothing fires, so the held turn would wait for
|
|
19
|
+
// the next event or for close. A timed flush is deliberately avoided — it would finalize turns off
|
|
20
|
+
// the caller's thread, and this state machine is only safe because every path runs on it.
|
|
21
|
+
const PARK_MAX_MS = 5000
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* @typedef {import('./turns').ResponseTurn} Turn
|
|
25
|
+
* @typedef {import('./turns').ToolCall} ToolCall
|
|
26
|
+
* @typedef {import('./turns').ToolResult} ToolResult
|
|
27
|
+
*
|
|
28
|
+
* @typedef {{
|
|
29
|
+
* startTime: number,
|
|
30
|
+
* finishTime: number,
|
|
31
|
+
* transcript: string,
|
|
32
|
+
* }} SpeechWindow
|
|
33
|
+
*
|
|
34
|
+
* @typedef {{
|
|
35
|
+
* text: string,
|
|
36
|
+
* transcript: string,
|
|
37
|
+
* audio: Buffer,
|
|
38
|
+
* audioPresent: boolean,
|
|
39
|
+
* mimeType: string,
|
|
40
|
+
* sampleRate: number,
|
|
41
|
+
* toolCalls?: ToolCall[],
|
|
42
|
+
* toolResults: ToolResult[],
|
|
43
|
+
* }} TurnSide
|
|
44
|
+
*
|
|
45
|
+
* @typedef {{
|
|
46
|
+
* sessionId: string,
|
|
47
|
+
* model: string | undefined,
|
|
48
|
+
* basePath: string,
|
|
49
|
+
* metadata: Record<string, unknown>,
|
|
50
|
+
* usage: object | undefined,
|
|
51
|
+
* failed: boolean,
|
|
52
|
+
* error: { type?: string, code?: string, message?: string } | undefined,
|
|
53
|
+
* runInContext: ((fn: () => void) => void) | undefined,
|
|
54
|
+
* root: { startTime: number, finishTime: number },
|
|
55
|
+
* llm: { startTime: number, finishTime: number },
|
|
56
|
+
* userSpeech: SpeechWindow | undefined,
|
|
57
|
+
* agentSpeech: SpeechWindow | undefined,
|
|
58
|
+
* input: TurnSide,
|
|
59
|
+
* output: TurnSide,
|
|
60
|
+
* }} TurnDescriptor
|
|
61
|
+
*/
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Coerce a value the server sent to a finite number, or `undefined`.
|
|
65
|
+
*
|
|
66
|
+
* @param {unknown} value
|
|
67
|
+
* @returns {number | undefined}
|
|
68
|
+
*/
|
|
69
|
+
function toFiniteNumber (value) {
|
|
70
|
+
const number = Number(value)
|
|
71
|
+
return Number.isFinite(number) ? number : undefined
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Flatten the provider's failure detail to plain strings, or `undefined` when it said nothing
|
|
76
|
+
* usable. Kept to the three fields OpenAI documents so a span carries the type and message a
|
|
77
|
+
* responder actually needs, without copying an arbitrary provider object onto a tag.
|
|
78
|
+
*
|
|
79
|
+
* @param {unknown} error
|
|
80
|
+
*/
|
|
81
|
+
function providerError (error) {
|
|
82
|
+
if (error === null || typeof error !== 'object') return
|
|
83
|
+
|
|
84
|
+
/** @type {{ type?: string, code?: string, message?: string }} */
|
|
85
|
+
const flattened = {}
|
|
86
|
+
let reported = false
|
|
87
|
+
for (const field of ['type', 'code', 'message']) {
|
|
88
|
+
if (error[field] == null) continue
|
|
89
|
+
flattened[field] = String(error[field])
|
|
90
|
+
reported = true
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
return reported ? flattened : undefined
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* The audio format to interpret a segment with: the one recorded when its first frame arrived,
|
|
98
|
+
* falling back to the session's current format.
|
|
99
|
+
*
|
|
100
|
+
* The fallback covers a segment that never recorded one — it holds no audio, or the session had not
|
|
101
|
+
* announced a format yet — and is not the mutable-format case: a segment that did record a format
|
|
102
|
+
* keeps it, so a later `session.update` cannot retime bytes that arrived under the old one.
|
|
103
|
+
*
|
|
104
|
+
* @param {import('./audio-accumulator')} audio
|
|
105
|
+
* @param {string} mimeType
|
|
106
|
+
* @param {number} sampleRate
|
|
107
|
+
* @returns {{ mimeType: string, sampleRate: number }}
|
|
108
|
+
*/
|
|
109
|
+
function segmentFormat (audio, mimeType, sampleRate) {
|
|
110
|
+
return { mimeType: audio.mimeType || mimeType, sampleRate: audio.sampleRate || sampleRate }
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Drives per-turn spans off one realtime connection's event stream.
|
|
115
|
+
*
|
|
116
|
+
* Turns are accumulated as plain data and replayed as a back-dated span tree at finalize, so nothing
|
|
117
|
+
* is held open across a deferred finalize (a late input transcription, or audio still playing) and a
|
|
118
|
+
* dropped connection cannot leak an unfinished span.
|
|
119
|
+
*/
|
|
120
|
+
class RealtimeSession {
|
|
121
|
+
/** Per-connection id grouping every turn of this conversation in the UI. */
|
|
122
|
+
#sessionId = randomUUID().replaceAll('-', '')
|
|
123
|
+
|
|
124
|
+
/** @type {(descriptor: TurnDescriptor) => void} */
|
|
125
|
+
#emitTurn
|
|
126
|
+
|
|
127
|
+
/** @type {(turn: Turn) => void} */
|
|
128
|
+
#captureContext
|
|
129
|
+
|
|
130
|
+
/** @type {string | undefined} */
|
|
131
|
+
#model
|
|
132
|
+
|
|
133
|
+
/** @type {string} */
|
|
134
|
+
#basePath
|
|
135
|
+
|
|
136
|
+
/** @type {Record<string, unknown>} */
|
|
137
|
+
#sessionConfig = {}
|
|
138
|
+
|
|
139
|
+
#inputAudioMime = ''
|
|
140
|
+
#outputAudioMime = ''
|
|
141
|
+
#inputAudioRate = DEFAULT_AUDIO_RATE
|
|
142
|
+
#outputAudioRate = DEFAULT_AUDIO_RATE
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Whether the session enabled input-audio transcription. A turn's finalize is only deferred to
|
|
146
|
+
* wait for a transcript when one is actually configured — otherwise none is ever coming.
|
|
147
|
+
*/
|
|
148
|
+
#inputTranscriptionEnabled = false
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Offset (ms) of the end of all input audio appended this session, and the wall clock (epoch ms)
|
|
152
|
+
* at which we reached it. Together they place a VAD event's buffer offset on the wall clock.
|
|
153
|
+
*/
|
|
154
|
+
#inputBufferMs = 0
|
|
155
|
+
/** @type {number | undefined} */
|
|
156
|
+
#inputBufferMsAt = undefined
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Bytes appended before the audio format was known — a client streaming from its own thread can
|
|
160
|
+
* beat `session.created`. Held as a backlog and folded into the clock once a rate arrives, rather
|
|
161
|
+
* than writing the origin off and disabling onset projection for the whole session.
|
|
162
|
+
*/
|
|
163
|
+
#pendingInputBytes = 0
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Whether anything consumes the captured audio, asked once per segment rather than held as a
|
|
167
|
+
* snapshot. Only LLM Observability does, so with it disabled — the default — the segments keep
|
|
168
|
+
* their byte counts, which size the speech windows, and hold no bytes. See `AudioAccumulator`.
|
|
169
|
+
*
|
|
170
|
+
* A predicate because the answer can change while a connection is open: `tracer.use('openai', {
|
|
171
|
+
* llmobs: false })` unsubscribes mid-session, and a snapshot taken at connect would keep buffering
|
|
172
|
+
* megabytes for a consumer that had gone away (or keep a session that gained one unable to
|
|
173
|
+
* capture). `AudioAccumulator` resolves it on the frame that opens a segment, which is as late as
|
|
174
|
+
* possible while still being once per segment: asking per frame could flip retention mid-clip and
|
|
175
|
+
* leave a segment holding only part of itself, while resolving at construction would be far too
|
|
176
|
+
* early — a pending input segment is built as soon as the previous turn starts.
|
|
177
|
+
*
|
|
178
|
+
* @type {() => boolean}
|
|
179
|
+
*/
|
|
180
|
+
#shouldRetainAudio
|
|
181
|
+
|
|
182
|
+
/** @type {InputTurn} */
|
|
183
|
+
#pendingInput
|
|
184
|
+
|
|
185
|
+
/** @type {Map<string, Turn>} */
|
|
186
|
+
#responses = new Map()
|
|
187
|
+
|
|
188
|
+
/** @type {Map<string, string>} */
|
|
189
|
+
#inputTranscripts = new Map()
|
|
190
|
+
|
|
191
|
+
/** call_id -> function name, so a later `function_call_output` can be labeled with its tool name. */
|
|
192
|
+
#toolCallNames = new Map()
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Turns whose response is done but whose input transcription hasn't arrived yet.
|
|
196
|
+
*
|
|
197
|
+
* @type {Turn[]}
|
|
198
|
+
*/
|
|
199
|
+
#awaiting = []
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Finished turns held open while their audio is still playing, so a barge-in truncation can still
|
|
203
|
+
* cap them.
|
|
204
|
+
*
|
|
205
|
+
* @type {Turn[]}
|
|
206
|
+
*/
|
|
207
|
+
#playing = []
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Whether this connection's client has ever truncated — which is what makes holding turns open
|
|
211
|
+
* worth its cost. A client either implements barge-in or it does not.
|
|
212
|
+
*/
|
|
213
|
+
#clientTruncates = false
|
|
214
|
+
|
|
215
|
+
#closed = false
|
|
216
|
+
|
|
217
|
+
/**
|
|
218
|
+
* @param {object} options
|
|
219
|
+
* @param {(descriptor: TurnDescriptor) => void} options.emitTurn
|
|
220
|
+
* @param {(turn: Turn) => void} options.captureContext
|
|
221
|
+
* @param {string} [options.model]
|
|
222
|
+
* @param {string} [options.basePath]
|
|
223
|
+
* @param {() => boolean} [options.shouldRetainAudio]
|
|
224
|
+
*/
|
|
225
|
+
constructor ({ emitTurn, captureContext, model, basePath = '', shouldRetainAudio = () => true }) {
|
|
226
|
+
this.#emitTurn = emitTurn
|
|
227
|
+
this.#captureContext = captureContext
|
|
228
|
+
this.#model = model
|
|
229
|
+
this.#basePath = basePath
|
|
230
|
+
this.#shouldRetainAudio = shouldRetainAudio
|
|
231
|
+
this.#pendingInput = new InputTurn(shouldRetainAudio)
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// -- event entry points ---------------------------------------------------
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* @param {Record<string, unknown>} event
|
|
238
|
+
* @param {number} now - Epoch ms at which the event was observed.
|
|
239
|
+
*/
|
|
240
|
+
onClientEvent (event, now) {
|
|
241
|
+
try {
|
|
242
|
+
this.#flushPlaying(now)
|
|
243
|
+
|
|
244
|
+
switch (event?.type) {
|
|
245
|
+
// First: a server-VAD client appends microphone audio many times a second.
|
|
246
|
+
case 'input_audio_buffer.append':
|
|
247
|
+
if (event.audio) this.#appendInputAudio(event.audio, now)
|
|
248
|
+
break
|
|
249
|
+
case 'conversation.item.truncate':
|
|
250
|
+
// Client -> server: "the listener only got this far into that item."
|
|
251
|
+
this.#onTruncate(event.item_id, event.audio_end_ms, now)
|
|
252
|
+
break
|
|
253
|
+
case 'session.update':
|
|
254
|
+
this.#updateSessionConfig(event.session)
|
|
255
|
+
break
|
|
256
|
+
case 'input_audio_buffer.clear':
|
|
257
|
+
// Discarded input audio must not be attributed to the next response.
|
|
258
|
+
this.#pendingInput.discardAudio()
|
|
259
|
+
break
|
|
260
|
+
case 'conversation.item.create':
|
|
261
|
+
this.#absorbInputItem(event.item, now)
|
|
262
|
+
break
|
|
263
|
+
}
|
|
264
|
+
} catch (error) {
|
|
265
|
+
log.debug('Error handling OpenAI realtime client event: %s', error?.message)
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* @param {Record<string, unknown>} event
|
|
271
|
+
* @param {number} now - Epoch ms at which the event was observed.
|
|
272
|
+
*/
|
|
273
|
+
onServerEvent (event, now) {
|
|
274
|
+
try {
|
|
275
|
+
this.#flushPlaying(now)
|
|
276
|
+
|
|
277
|
+
const eventType = event?.type
|
|
278
|
+
if (typeof eventType !== 'string') return
|
|
279
|
+
|
|
280
|
+
switch (eventType) {
|
|
281
|
+
case 'session.created':
|
|
282
|
+
case 'session.updated':
|
|
283
|
+
this.#updateSessionConfig(event.session)
|
|
284
|
+
return
|
|
285
|
+
case 'conversation.item.truncated':
|
|
286
|
+
// The server's acknowledgement of a client truncation. `capTo` is absolute, so handling
|
|
287
|
+
// both it and the client event applies the cap once.
|
|
288
|
+
this.#onTruncate(event.item_id, event.audio_end_ms, now)
|
|
289
|
+
return
|
|
290
|
+
case 'input_audio_buffer.speech_started':
|
|
291
|
+
this.#onSpeechStarted(event.audio_start_ms, now)
|
|
292
|
+
return
|
|
293
|
+
case 'input_audio_buffer.speech_stopped':
|
|
294
|
+
// The commit that follows is the authoritative end of user speech and overwrites this;
|
|
295
|
+
// recording it here only covers a session that never commits, so the window still gets an
|
|
296
|
+
// end.
|
|
297
|
+
this.#pendingInput.speechEndTime ??= now
|
|
298
|
+
return
|
|
299
|
+
case 'input_audio_buffer.committed':
|
|
300
|
+
this.#pendingInput.itemId = event.item_id == null ? undefined : String(event.item_id)
|
|
301
|
+
this.#pendingInput.speechEndTime = now
|
|
302
|
+
return
|
|
303
|
+
case 'input_audio_buffer.cleared':
|
|
304
|
+
this.#pendingInput.discardAudio()
|
|
305
|
+
return
|
|
306
|
+
case 'conversation.item.input_audio_transcription.completed':
|
|
307
|
+
this.#onInputTranscript(event.item_id, event.transcript, now)
|
|
308
|
+
return
|
|
309
|
+
case 'conversation.item.input_audio_transcription.failed':
|
|
310
|
+
// No transcription is coming for this item. Recorded as an empty transcript rather than
|
|
311
|
+
// just flushing whatever is already waiting: this can land *before* the response's
|
|
312
|
+
// `response.done`, and without the record `#finishResponse` would then park the turn for
|
|
313
|
+
// an event that has already happened.
|
|
314
|
+
this.#onInputTranscript(event.item_id, '', now)
|
|
315
|
+
return
|
|
316
|
+
case 'response.created':
|
|
317
|
+
this.#startResponse(event.response?.id ?? event.response_id, event.response, now)
|
|
318
|
+
return
|
|
319
|
+
case 'response.done':
|
|
320
|
+
this.#finishResponse(event.response?.id ?? event.response_id, event.response, now)
|
|
321
|
+
return
|
|
322
|
+
default:
|
|
323
|
+
this.#handleResponseDelta(event, eventType, now)
|
|
324
|
+
}
|
|
325
|
+
} catch (error) {
|
|
326
|
+
log.debug('Error handling OpenAI realtime server event: %s', error?.message)
|
|
327
|
+
}
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
/**
|
|
331
|
+
* Finalize everything still open. Idempotent, so every close path can call it.
|
|
332
|
+
*
|
|
333
|
+
* @param {number} now - Epoch ms.
|
|
334
|
+
* @param {boolean} [failed] - The connection ended abnormally. Only responses still in flight are
|
|
335
|
+
* marked: a turn already awaiting a transcript or playback had its `response.done`, so it
|
|
336
|
+
* succeeded whatever the transport did afterwards.
|
|
337
|
+
*/
|
|
338
|
+
finishSession (now, failed = false) {
|
|
339
|
+
if (this.#closed) return
|
|
340
|
+
this.#closed = true
|
|
341
|
+
|
|
342
|
+
try {
|
|
343
|
+
this.#flushAwaiting(now)
|
|
344
|
+
this.#flushPlaying(now, true)
|
|
345
|
+
|
|
346
|
+
// In-flight turns that never saw `response.done` (closed mid-turn). Whatever partial data we
|
|
347
|
+
// have is submitted.
|
|
348
|
+
for (const turn of this.#responses.values()) {
|
|
349
|
+
if (failed) turn.status = 'failed'
|
|
350
|
+
this.#applyCachedTranscript(turn)
|
|
351
|
+
this.#finalizeTurn(turn, now, true)
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
this.#responses.clear()
|
|
355
|
+
this.#inputTranscripts.clear()
|
|
356
|
+
this.#toolCallNames.clear()
|
|
357
|
+
// Audio buffered for a turn that will now never start. An app that holds on to closed
|
|
358
|
+
// transport objects keeps their sessions reachable through the connection map, and a
|
|
359
|
+
// server-VAD client streams the microphone continuously, so this is up to the whole retention
|
|
360
|
+
// cap per closed connection with nothing left to consume it.
|
|
361
|
+
this.#pendingInput = new InputTurn(this.#shouldRetainAudio)
|
|
362
|
+
} catch (error) {
|
|
363
|
+
log.debug('Error finalizing OpenAI realtime session: %s', error?.message)
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// -- response deltas ------------------------------------------------------
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* @param {Record<string, unknown>} event
|
|
371
|
+
* @param {string} eventType
|
|
372
|
+
* @param {number} now
|
|
373
|
+
*/
|
|
374
|
+
#handleResponseDelta (event, eventType, now) {
|
|
375
|
+
const turn = this.#responses.get(String(event.response_id))
|
|
376
|
+
if (turn === undefined) return
|
|
377
|
+
|
|
378
|
+
switch (normalizeResponseEventType(eventType)) {
|
|
379
|
+
case 'response.audio.delta': {
|
|
380
|
+
const { delta, item_id: itemId } = event
|
|
381
|
+
if (!delta) return
|
|
382
|
+
if (itemId != null) {
|
|
383
|
+
// Remember where this item's audio starts in the turn's segment, before appending, so a
|
|
384
|
+
// truncation reported against the item maps onto the segment.
|
|
385
|
+
const key = String(itemId)
|
|
386
|
+
if (!turn.audioItemStarts.has(key)) turn.audioItemStarts.set(key, turn.audio.totalDecodedBytes)
|
|
387
|
+
}
|
|
388
|
+
turn.audio.append(delta, now, this.#outputAudioMime, this.#outputAudioRate)
|
|
389
|
+
return
|
|
390
|
+
}
|
|
391
|
+
case 'response.audio_transcript.delta':
|
|
392
|
+
turn.transcript.appendDelta(event.item_id, event.delta ?? '')
|
|
393
|
+
return
|
|
394
|
+
case 'response.audio_transcript.done':
|
|
395
|
+
if (event.transcript != null) turn.transcript.complete(event.item_id, String(event.transcript))
|
|
396
|
+
return
|
|
397
|
+
case 'response.text.delta':
|
|
398
|
+
turn.text.appendDelta(event.item_id, event.delta ?? '')
|
|
399
|
+
return
|
|
400
|
+
case 'response.text.done':
|
|
401
|
+
if (event.text != null) turn.text.complete(event.item_id, String(event.text))
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// -- input audio and the speech window ------------------------------------
|
|
406
|
+
|
|
407
|
+
/**
|
|
408
|
+
* Buffer a client audio append for the pending turn and advance the input-buffer clock.
|
|
409
|
+
*
|
|
410
|
+
* @param {string} base64
|
|
411
|
+
* @param {number} now
|
|
412
|
+
*/
|
|
413
|
+
#appendInputAudio (base64, now) {
|
|
414
|
+
const pending = this.#pendingInput
|
|
415
|
+
|
|
416
|
+
// Fold in anything buffered before the format was known first, so the base offset captured below
|
|
417
|
+
// sits on the same timeline the VAD offsets are later projected against.
|
|
418
|
+
this.#advanceInputBufferClock(0, now)
|
|
419
|
+
|
|
420
|
+
if (pending.audio.startTime === undefined && this.#inputBufferMsAt !== undefined) {
|
|
421
|
+
// First frame of this turn and the clock is live: remember where it sits on the session's
|
|
422
|
+
// input-buffer timeline, so a VAD offset can be turned into a byte offset into what we buffer
|
|
423
|
+
// here. Left unset while the clock is dead, since a base of 0 would read as "this turn starts
|
|
424
|
+
// at the session origin" and over-trim the front of the segment.
|
|
425
|
+
pending.audioBaseMs = this.#inputBufferMs
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
const decodedBytes = pending.audio.append(base64, now, this.#inputAudioMime, this.#inputAudioRate)
|
|
429
|
+
this.#advanceInputBufferClock(decodedBytes, now)
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
/**
|
|
433
|
+
* Track how far into the session's input audio the buffer now extends, and when we got there.
|
|
434
|
+
*
|
|
435
|
+
* @param {number} decodedBytes
|
|
436
|
+
* @param {number} now
|
|
437
|
+
*/
|
|
438
|
+
#advanceInputBufferClock (decodedBytes, now) {
|
|
439
|
+
this.#pendingInputBytes += decodedBytes
|
|
440
|
+
|
|
441
|
+
const rate = bytesPerSecond(this.#inputAudioMime, this.#inputAudioRate)
|
|
442
|
+
// A format we can never rate leaves the projection unavailable: the backlog simply never
|
|
443
|
+
// converts and `#inputBufferMsAt` stays undefined, which is what marks the clock dead.
|
|
444
|
+
if (!rate) return
|
|
445
|
+
|
|
446
|
+
this.#inputBufferMs += this.#pendingInputBytes / rate * 1000
|
|
447
|
+
this.#pendingInputBytes = 0
|
|
448
|
+
this.#inputBufferMsAt = now
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
/**
|
|
452
|
+
* Anchor the pending turn's user-speech window on the VAD speech onset.
|
|
453
|
+
*
|
|
454
|
+
* A server-VAD client streams the microphone continuously, so the first buffer append of a turn
|
|
455
|
+
* lands the instant the *previous* turn was committed: it marks when we started listening, not
|
|
456
|
+
* when the human started speaking. Left at that, every user-speech window swallows the whole
|
|
457
|
+
* preceding agent response and consecutive turns overlap on the session timeline.
|
|
458
|
+
* `input_audio_buffer.speech_started` is the real onset, and the audio it points at
|
|
459
|
+
* (`audio_start_ms`, which already includes the session's `prefix_padding_ms`) is the audio worth
|
|
460
|
+
* keeping, so the buffered lead-in is trimmed off the front to keep the captured audio and the
|
|
461
|
+
* reported window in step.
|
|
462
|
+
*
|
|
463
|
+
* Only the first onset of a turn counts: a turn that VAD splits into several speech runs before a
|
|
464
|
+
* single commit is still one committed item, which began at the first run.
|
|
465
|
+
*
|
|
466
|
+
* @param {unknown} audioStartMs
|
|
467
|
+
* @param {number} now
|
|
468
|
+
*/
|
|
469
|
+
#onSpeechStarted (audioStartMs, now) {
|
|
470
|
+
const pending = this.#pendingInput
|
|
471
|
+
if (pending.speechStartTime !== undefined) return
|
|
472
|
+
|
|
473
|
+
const onset = this.#bufferOffsetToWallTime(audioStartMs, now)
|
|
474
|
+
pending.speechStartTime = onset
|
|
475
|
+
|
|
476
|
+
pending.audio.trimLeading(this.#preOnsetBytes(audioStartMs))
|
|
477
|
+
// Re-anchor the segment on the onset: whatever survived the trim starts there. A trim that
|
|
478
|
+
// covered the whole segment resets it instead, and re-anchoring then would leave `startTime`
|
|
479
|
+
// set with no format recorded — `append` only records the format on the frame that opens a
|
|
480
|
+
// segment, so every later frame would skip it and the segment would fall back to whatever
|
|
481
|
+
// format the session holds at describe time.
|
|
482
|
+
if (pending.audio.startTime !== undefined) pending.audio.startTime = onset
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/**
|
|
486
|
+
* Byte count of the audio buffered for this turn ahead of the speech onset.
|
|
487
|
+
*
|
|
488
|
+
* @param {unknown} audioStartMs
|
|
489
|
+
*/
|
|
490
|
+
#preOnsetBytes (audioStartMs) {
|
|
491
|
+
const baseMs = this.#pendingInput.audioBaseMs
|
|
492
|
+
const onsetMs = toFiniteNumber(audioStartMs)
|
|
493
|
+
const rate = bytesPerSecond(this.#inputAudioMime, this.#inputAudioRate)
|
|
494
|
+
if (baseMs === undefined || onsetMs === undefined || !rate) return 0
|
|
495
|
+
|
|
496
|
+
const bytes = Math.max(0, Math.trunc((onsetMs - baseMs) / 1000 * rate))
|
|
497
|
+
// Cut on a sample boundary, as the truncation cap does: `audioBaseMs` is a fractional millisecond
|
|
498
|
+
// offset, so this lands on an odd byte often enough to matter, and slicing a PCM16 segment
|
|
499
|
+
// mid-sample pairs every low byte with the next sample's high byte — the whole clip decodes to
|
|
500
|
+
// noise. Rounding down keeps a byte of lead-in rather than corrupting what follows.
|
|
501
|
+
return bytes - bytes % 2 // keep PCM16 samples whole; a byte is nothing for G.711
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
/**
|
|
505
|
+
* Project an input-buffer offset — a VAD event's `audio_start_ms`, measured from the start of all
|
|
506
|
+
* audio written to the buffer this session — onto the wall clock.
|
|
507
|
+
*
|
|
508
|
+
* This assumes the client appends audio roughly in real time, which is true for a live microphone,
|
|
509
|
+
* the only case where a wall-clock speech window means anything. Knowing how much audio had been
|
|
510
|
+
* appended at a known instant, the offset is that instant minus the audio still ahead of it. The
|
|
511
|
+
* result is clamped to a window we can defend — no earlier than this turn's first buffered frame,
|
|
512
|
+
* no later than when we observed the event — so a bursty or pre-recorded sender degrades to a sane
|
|
513
|
+
* bound instead of a wild timestamp, and falls back to the observation time when the projection is
|
|
514
|
+
* unavailable.
|
|
515
|
+
*
|
|
516
|
+
* @param {unknown} offsetMs
|
|
517
|
+
* @param {number} observedTime
|
|
518
|
+
*/
|
|
519
|
+
#bufferOffsetToWallTime (offsetMs, observedTime) {
|
|
520
|
+
const offset = toFiniteNumber(offsetMs)
|
|
521
|
+
if (offset === undefined || this.#inputBufferMsAt === undefined) return observedTime
|
|
522
|
+
|
|
523
|
+
let projected = this.#inputBufferMsAt - (this.#inputBufferMs - offset)
|
|
524
|
+
|
|
525
|
+
const earliest = this.#pendingInput.audio.startTime
|
|
526
|
+
if (earliest !== undefined) projected = Math.max(projected, earliest)
|
|
527
|
+
|
|
528
|
+
return Math.min(projected, observedTime)
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
// -- barge-in (agent playback cut short) ----------------------------------
|
|
532
|
+
|
|
533
|
+
/**
|
|
534
|
+
* Cap an assistant audio segment at what the listener actually heard.
|
|
535
|
+
*
|
|
536
|
+
* Over a WebSocket the client owns playback and the model streams audio faster than it plays, so
|
|
537
|
+
* on a barge-in the client stops its speaker and reports how far it got. Audio delivered past that
|
|
538
|
+
* point was never heard; without this the stored agent audio — and the agent-speech window derived
|
|
539
|
+
* from it — covers the whole generated response and runs past the interruption into the next user
|
|
540
|
+
* turn.
|
|
541
|
+
*
|
|
542
|
+
* Seeing a truncation also marks this connection's client as one that cuts playback short, which
|
|
543
|
+
* is what makes holding its turns open worthwhile.
|
|
544
|
+
*
|
|
545
|
+
* @param {unknown} itemId
|
|
546
|
+
* @param {unknown} audioEndMs
|
|
547
|
+
* @param {number} now
|
|
548
|
+
*/
|
|
549
|
+
#onTruncate (itemId, audioEndMs, now) {
|
|
550
|
+
this.#clientTruncates = true
|
|
551
|
+
|
|
552
|
+
const endMs = toFiniteNumber(audioEndMs)
|
|
553
|
+
if (itemId == null || endMs === undefined) return
|
|
554
|
+
|
|
555
|
+
const item = String(itemId)
|
|
556
|
+
for (const turn of this.#openTurns()) {
|
|
557
|
+
const itemStart = turn.audioItemStarts.get(item)
|
|
558
|
+
if (itemStart === undefined) continue
|
|
559
|
+
|
|
560
|
+
const { mimeType, sampleRate } = segmentFormat(turn.audio, this.#outputAudioMime, this.#outputAudioRate)
|
|
561
|
+
const rate = bytesPerSecond(mimeType, sampleRate)
|
|
562
|
+
if (!rate) return
|
|
563
|
+
|
|
564
|
+
const cap = itemStart + Math.trunc(endMs / 1000 * rate)
|
|
565
|
+
turn.audio.capTo(cap - cap % 2) // keep PCM16 samples whole; a byte is nothing for G.711
|
|
566
|
+
|
|
567
|
+
const playingIndex = this.#playing.indexOf(turn)
|
|
568
|
+
if (playingIndex !== -1) {
|
|
569
|
+
// Playback ended when the listener cut it off, so stop waiting on it.
|
|
570
|
+
this.#playing.splice(playingIndex, 1)
|
|
571
|
+
this.#finalizeTurn(turn, now, true)
|
|
572
|
+
}
|
|
573
|
+
return
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
/**
|
|
578
|
+
* Every turn we could still amend: in flight, awaiting a transcript, or awaiting playback.
|
|
579
|
+
*
|
|
580
|
+
* @returns {Turn[]}
|
|
581
|
+
*/
|
|
582
|
+
#openTurns () {
|
|
583
|
+
return [...this.#responses.values(), ...this.#awaiting, ...this.#playing]
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
/**
|
|
587
|
+
* Hold a finished turn while its audio is still playing, so a late truncation can still cap it.
|
|
588
|
+
*
|
|
589
|
+
* `response.done` normally lands mid-playback (generation outruns playback), and a barge-in
|
|
590
|
+
* truncation arrives after that — too late for a turn we already submitted. Holding costs
|
|
591
|
+
* submission latency and a wider window in which to lose the turn if the process dies, so we only
|
|
592
|
+
* hold on connections whose client has actually truncated before. Clients that never truncate hear
|
|
593
|
+
* every byte we captured and finalize at `response.done` exactly as before, paying nothing. The
|
|
594
|
+
* cost of that trade is that the first interruption on a connection is reported untruncated.
|
|
595
|
+
*
|
|
596
|
+
* Reports whether the turn was parked, so the caller knows not to finalize it.
|
|
597
|
+
*
|
|
598
|
+
* @param {Turn} turn
|
|
599
|
+
* @param {number} now
|
|
600
|
+
*/
|
|
601
|
+
#parkForPlayback (turn, now) {
|
|
602
|
+
if (!this.#clientTruncates || turn.audio.startTime === undefined) return false
|
|
603
|
+
|
|
604
|
+
const { mimeType, sampleRate } = segmentFormat(turn.audio, this.#outputAudioMime, this.#outputAudioRate)
|
|
605
|
+
const playbackMs = segmentDurationMs(turn.audio.totalDecodedBytes, mimeType, sampleRate)
|
|
606
|
+
if (playbackMs === undefined) return false
|
|
607
|
+
|
|
608
|
+
const endTime = turn.audio.startTime + playbackMs
|
|
609
|
+
if (endTime <= now) return false
|
|
610
|
+
|
|
611
|
+
turn.playbackEndTime = Math.min(endTime, now + PARK_MAX_MS)
|
|
612
|
+
this.#playing.push(turn)
|
|
613
|
+
return true
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
/**
|
|
617
|
+
* Finalize held turns whose audio has finished playing, or all of them when forced.
|
|
618
|
+
*
|
|
619
|
+
* Event-driven rather than timed: a realtime connection is chatty — a streaming client appends
|
|
620
|
+
* microphone audio continuously — so this runs often enough to submit a turn shortly after its
|
|
621
|
+
* playback ends, and the next turn and connection close both force it so a turn can't leak.
|
|
622
|
+
*
|
|
623
|
+
* @param {number} now
|
|
624
|
+
* @param {boolean} [force]
|
|
625
|
+
*/
|
|
626
|
+
#flushPlaying (now, force = false) {
|
|
627
|
+
if (this.#playing.length === 0) return
|
|
628
|
+
|
|
629
|
+
for (let i = this.#playing.length - 1; i >= 0; i--) {
|
|
630
|
+
const turn = this.#playing[i]
|
|
631
|
+
if (!force && turn.playbackEndTime !== undefined && turn.playbackEndTime > now) continue
|
|
632
|
+
|
|
633
|
+
this.#playing.splice(i, 1)
|
|
634
|
+
this.#finalizeTurn(turn, now, true)
|
|
635
|
+
}
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
// -- turn lifecycle -------------------------------------------------------
|
|
639
|
+
|
|
640
|
+
/**
|
|
641
|
+
* @param {unknown} responseId
|
|
642
|
+
* @param {Record<string, unknown> | undefined} response - The created response, which says which
|
|
643
|
+
* conversation it belongs to.
|
|
644
|
+
* @param {number} now
|
|
645
|
+
*/
|
|
646
|
+
#startResponse (responseId, response, now) {
|
|
647
|
+
if (responseId == null) return
|
|
648
|
+
|
|
649
|
+
// The server tells us directly: a response created with `conversation: 'none'` reports a null
|
|
650
|
+
// `conversation_id`, while one in the default conversation — including every response server VAD
|
|
651
|
+
// creates on its own — reports an id. Only the latter owns the buffered user input.
|
|
652
|
+
//
|
|
653
|
+
// Read off `response.created` rather than correlated back to the client's `response.create`.
|
|
654
|
+
// Position-matching the two streams cannot survive a create the server rejects (no
|
|
655
|
+
// `response.created` ever arrives) or several in flight at once, and both of those
|
|
656
|
+
// mis-classifications cost a real turn its user speech.
|
|
657
|
+
//
|
|
658
|
+
// `conversation_id` is absent rather than null on the beta realtime surface, which is why this
|
|
659
|
+
// tests for null exactly: an unknown conversation is treated as the conversation, matching how
|
|
660
|
+
// that surface behaved before out-of-band responses were modelled at all.
|
|
661
|
+
const outOfBand = response?.conversation_id === null
|
|
662
|
+
|
|
663
|
+
// Classify before flushing: an out-of-band response runs *alongside* the conversation, so it is
|
|
664
|
+
// not evidence that a pending turn's transcript has stopped coming.
|
|
665
|
+
|
|
666
|
+
if (!outOfBand) {
|
|
667
|
+
// A new conversational turn starting means a prior turn's input transcription is almost
|
|
668
|
+
// certainly not coming anymore, and that any held playback is over (or was cut off without a
|
|
669
|
+
// truncation reaching us), so flush both rather than let a turn hang.
|
|
670
|
+
this.#flushAwaiting(now)
|
|
671
|
+
this.#flushPlaying(now, true)
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
const input = outOfBand ? new InputTurn(this.#shouldRetainAudio) : this.#pendingInput
|
|
675
|
+
const turn = new ResponseTurn(input, now, this.#shouldRetainAudio)
|
|
676
|
+
if (!outOfBand) this.#pendingInput = new InputTurn(this.#shouldRetainAudio)
|
|
677
|
+
turn.model = this.#model
|
|
678
|
+
|
|
679
|
+
this.#captureContext(turn)
|
|
680
|
+
this.#responses.set(String(responseId), turn)
|
|
681
|
+
}
|
|
682
|
+
|
|
683
|
+
/**
|
|
684
|
+
* @param {unknown} responseId
|
|
685
|
+
* @param {Record<string, unknown>} response
|
|
686
|
+
* @param {number} now
|
|
687
|
+
*/
|
|
688
|
+
#finishResponse (responseId, response, now) {
|
|
689
|
+
if (responseId == null) return
|
|
690
|
+
|
|
691
|
+
const key = String(responseId)
|
|
692
|
+
const turn = this.#responses.get(key)
|
|
693
|
+
if (turn === undefined) return
|
|
694
|
+
this.#responses.delete(key)
|
|
695
|
+
|
|
696
|
+
turn.responseDoneTime = now
|
|
697
|
+
turn.usage = response?.usage
|
|
698
|
+
turn.model = response?.model || turn.model || this.#model
|
|
699
|
+
turn.status = response?.status
|
|
700
|
+
// The provider's actionable detail for a failure. Without it a failed turn reaches the backend
|
|
701
|
+
// as a bare `error: 1`, which says a realtime call broke but nothing about why.
|
|
702
|
+
turn.error = providerError(response?.status_details?.error)
|
|
703
|
+
|
|
704
|
+
const { toolCalls, toolResults } = extractResponseTools(response)
|
|
705
|
+
turn.toolCalls = toolCalls
|
|
706
|
+
turn.toolResults = toolResults
|
|
707
|
+
// Remember each function call's name so the `function_call_output` the app returns later can be
|
|
708
|
+
// labeled with it — the output event itself only carries the call_id.
|
|
709
|
+
for (const toolCall of toolCalls) {
|
|
710
|
+
if (toolCall.type === 'function' && toolCall.toolId) {
|
|
711
|
+
this.#toolCallNames.set(toolCall.toolId, toolCall.name)
|
|
712
|
+
}
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
this.#applyCachedTranscript(turn)
|
|
716
|
+
|
|
717
|
+
// Hold the turn open for a late input transcription ONLY when transcription is actually enabled
|
|
718
|
+
// and this item has not already reached a terminal state: otherwise no transcript is ever
|
|
719
|
+
// coming, and waiting would needlessly delay every turn until the next one, and the last turn
|
|
720
|
+
// until close. A cached entry — including the empty string a completion-with-no-text or a
|
|
721
|
+
// failure records — is that terminal state, and it is why the check is `has` rather than the
|
|
722
|
+
// transcript's truthiness.
|
|
723
|
+
if (!turn.input.transcript &&
|
|
724
|
+
turn.input.itemId !== undefined &&
|
|
725
|
+
this.#inputTranscriptionEnabled &&
|
|
726
|
+
!this.#inputTranscripts.has(turn.input.itemId)) {
|
|
727
|
+
this.#awaiting.push(turn)
|
|
728
|
+
return
|
|
729
|
+
}
|
|
730
|
+
|
|
731
|
+
this.#finalizeTurn(turn, now)
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
/**
|
|
735
|
+
* @param {unknown} itemId
|
|
736
|
+
* @param {unknown} transcript
|
|
737
|
+
* @param {number} now
|
|
738
|
+
*/
|
|
739
|
+
#onInputTranscript (itemId, transcript, now) {
|
|
740
|
+
const item = itemId == null ? undefined : String(itemId)
|
|
741
|
+
const text = transcript == null ? '' : String(transcript)
|
|
742
|
+
|
|
743
|
+
if (item !== undefined) this.#inputTranscripts.set(item, text)
|
|
744
|
+
if (this.#pendingInput.itemId === item && !this.#pendingInput.transcript) {
|
|
745
|
+
this.#pendingInput.transcript = text
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
// A finished turn may have been waiting on exactly this transcript — finalize it now.
|
|
749
|
+
this.#finalizeAwaitingFor(itemId, now, text)
|
|
750
|
+
}
|
|
751
|
+
|
|
752
|
+
/**
|
|
753
|
+
* @param {unknown} itemId
|
|
754
|
+
* @param {number} now
|
|
755
|
+
* @param {string} [transcript]
|
|
756
|
+
*/
|
|
757
|
+
#finalizeAwaitingFor (itemId, now, transcript) {
|
|
758
|
+
if (itemId == null) return
|
|
759
|
+
const item = String(itemId)
|
|
760
|
+
|
|
761
|
+
for (let i = this.#awaiting.length - 1; i >= 0; i--) {
|
|
762
|
+
const turn = this.#awaiting[i]
|
|
763
|
+
if (turn.input.itemId !== item) continue
|
|
764
|
+
|
|
765
|
+
if (transcript) turn.input.transcript ||= transcript
|
|
766
|
+
this.#awaiting.splice(i, 1)
|
|
767
|
+
this.#finalizeTurn(turn, now)
|
|
768
|
+
}
|
|
769
|
+
}
|
|
770
|
+
|
|
771
|
+
/**
|
|
772
|
+
* @param {number} now
|
|
773
|
+
*/
|
|
774
|
+
#flushAwaiting (now) {
|
|
775
|
+
if (this.#awaiting.length === 0) return
|
|
776
|
+
|
|
777
|
+
const awaiting = this.#awaiting
|
|
778
|
+
this.#awaiting = []
|
|
779
|
+
for (const turn of awaiting) this.#finalizeTurn(turn, now, true)
|
|
780
|
+
}
|
|
781
|
+
|
|
782
|
+
/**
|
|
783
|
+
* @param {Turn} turn
|
|
784
|
+
*/
|
|
785
|
+
#applyCachedTranscript (turn) {
|
|
786
|
+
if (turn.input.transcript || turn.input.itemId === undefined) return
|
|
787
|
+
turn.input.transcript = this.#inputTranscripts.get(turn.input.itemId) ?? ''
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
/**
|
|
791
|
+
* @param {Turn} turn
|
|
792
|
+
* @param {number} now
|
|
793
|
+
* @param {boolean} [force] - Skip parking, for the paths that must not wait (close, next turn).
|
|
794
|
+
*/
|
|
795
|
+
#finalizeTurn (turn, now, force = false) {
|
|
796
|
+
// The turn's data is complete, but on a barge-in-capable client the agent's audio may still be
|
|
797
|
+
// playing and a truncation may yet cut it short — hold the turn rather than submit audio the
|
|
798
|
+
// listener might never hear.
|
|
799
|
+
if (!force && this.#parkForPlayback(turn, now)) return
|
|
800
|
+
|
|
801
|
+
// Drop the cached transcript for this turn's input item so the map can't grow across a long
|
|
802
|
+
// session. Every finalize path goes through here.
|
|
803
|
+
if (turn.input.itemId !== undefined) this.#inputTranscripts.delete(turn.input.itemId)
|
|
804
|
+
|
|
805
|
+
try {
|
|
806
|
+
this.#emitTurn(this.#describeTurn(turn, now))
|
|
807
|
+
} catch (error) {
|
|
808
|
+
log.debug('Error emitting OpenAI realtime turn spans: %s', error?.message)
|
|
809
|
+
}
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
/**
|
|
813
|
+
* Flatten a finished turn into the boundaries and payloads the plugins need, so they never reach
|
|
814
|
+
* into this class's state.
|
|
815
|
+
*
|
|
816
|
+
* @param {Turn} turn
|
|
817
|
+
* @param {number} now
|
|
818
|
+
* @returns {TurnDescriptor}
|
|
819
|
+
*/
|
|
820
|
+
#describeTurn (turn, now) {
|
|
821
|
+
const { input } = turn
|
|
822
|
+
|
|
823
|
+
// The VAD speech onset when we have it. The first buffered frame only approximates the onset,
|
|
824
|
+
// for a client that appends audio solely while the user talks (client-side turn detection).
|
|
825
|
+
const userStart = input.speechStartTime ?? input.audio.startTime
|
|
826
|
+
const inputFormat = segmentFormat(input.audio, this.#inputAudioMime, this.#inputAudioRate)
|
|
827
|
+
const outputFormat = segmentFormat(turn.audio, this.#outputAudioMime, this.#outputAudioRate)
|
|
828
|
+
const inputDurationMs = segmentDurationMs(
|
|
829
|
+
input.audio.totalDecodedBytes, inputFormat.mimeType, inputFormat.sampleRate
|
|
830
|
+
)
|
|
831
|
+
const userEnd = input.speechEndTime ??
|
|
832
|
+
(userStart !== undefined && inputDurationMs !== undefined ? userStart + inputDurationMs : undefined)
|
|
833
|
+
|
|
834
|
+
const agentStart = turn.audio.startTime
|
|
835
|
+
const playbackMs = segmentDurationMs(
|
|
836
|
+
turn.audio.totalDecodedBytes, outputFormat.mimeType, outputFormat.sampleRate
|
|
837
|
+
)
|
|
838
|
+
const agentEnd = agentStart === undefined
|
|
839
|
+
? undefined
|
|
840
|
+
: (playbackMs === undefined ? (turn.responseDoneTime ?? now) : agentStart + playbackMs)
|
|
841
|
+
|
|
842
|
+
// The llm span measures model work: it opens at the end of user speech, not when the human
|
|
843
|
+
// started talking, and closes when generation completes.
|
|
844
|
+
const llmStartTime = input.speechEndTime ?? input.audio.startTime ?? turn.createdTime
|
|
845
|
+
const llmFinishTime = Math.max(turn.responseDoneTime ?? now, llmStartTime)
|
|
846
|
+
|
|
847
|
+
const rootStartTime = userStart ?? input.speechEndTime ?? turn.createdTime
|
|
848
|
+
// Unlike dd-trace-py, clamp the root to its children's actual ends so it always contains them: a
|
|
849
|
+
// turn finalized without `response.done` would otherwise finish before its own llm span.
|
|
850
|
+
const rootFinishTime = Math.max(rootStartTime, llmFinishTime, agentEnd ?? 0, userEnd ?? 0)
|
|
851
|
+
|
|
852
|
+
return {
|
|
853
|
+
sessionId: this.#sessionId,
|
|
854
|
+
model: turn.model,
|
|
855
|
+
basePath: this.#basePath,
|
|
856
|
+
metadata: { ...this.#sessionConfig },
|
|
857
|
+
usage: turn.usage,
|
|
858
|
+
failed: turn.status === 'failed',
|
|
859
|
+
error: turn.error,
|
|
860
|
+
runInContext: turn.runInContext,
|
|
861
|
+
root: { startTime: rootStartTime, finishTime: rootFinishTime },
|
|
862
|
+
llm: { startTime: llmStartTime, finishTime: llmFinishTime },
|
|
863
|
+
userSpeech: userStart === undefined || userEnd === undefined
|
|
864
|
+
? undefined
|
|
865
|
+
: {
|
|
866
|
+
startTime: userStart,
|
|
867
|
+
finishTime: Math.max(userStart, userEnd),
|
|
868
|
+
transcript: input.transcript || input.text,
|
|
869
|
+
},
|
|
870
|
+
agentSpeech: agentStart === undefined || agentEnd === undefined
|
|
871
|
+
? undefined
|
|
872
|
+
: {
|
|
873
|
+
startTime: agentStart,
|
|
874
|
+
finishTime: Math.max(agentStart, agentEnd),
|
|
875
|
+
transcript: turn.transcript.value || turn.text.value,
|
|
876
|
+
},
|
|
877
|
+
input: {
|
|
878
|
+
text: input.text,
|
|
879
|
+
transcript: input.transcript,
|
|
880
|
+
audio: input.audio.toBuffer(),
|
|
881
|
+
audioPresent: input.audio.present,
|
|
882
|
+
mimeType: inputFormat.mimeType,
|
|
883
|
+
sampleRate: inputFormat.sampleRate,
|
|
884
|
+
toolResults: input.toolResults,
|
|
885
|
+
},
|
|
886
|
+
output: {
|
|
887
|
+
text: turn.text.value,
|
|
888
|
+
transcript: turn.transcript.value,
|
|
889
|
+
audio: turn.audio.toBuffer(),
|
|
890
|
+
audioPresent: turn.audio.present,
|
|
891
|
+
mimeType: outputFormat.mimeType,
|
|
892
|
+
sampleRate: outputFormat.sampleRate,
|
|
893
|
+
toolCalls: turn.toolCalls,
|
|
894
|
+
toolResults: turn.toolResults,
|
|
895
|
+
},
|
|
896
|
+
}
|
|
897
|
+
}
|
|
898
|
+
|
|
899
|
+
// -- config extraction ----------------------------------------------------
|
|
900
|
+
|
|
901
|
+
/**
|
|
902
|
+
* @param {Record<string, unknown>} session
|
|
903
|
+
*/
|
|
904
|
+
#updateSessionConfig (session) {
|
|
905
|
+
if (session == null) return
|
|
906
|
+
|
|
907
|
+
if (typeof session.model === 'string' && session.model) this.#model = session.model
|
|
908
|
+
if (session.instructions != null) this.#sessionConfig.instructions = String(session.instructions)
|
|
909
|
+
|
|
910
|
+
const modalities = session.output_modalities ?? session.modalities
|
|
911
|
+
if (modalities) this.#sessionConfig.output_modalities = [...modalities]
|
|
912
|
+
|
|
913
|
+
const { audio } = session
|
|
914
|
+
let inputFormat = audio?.input?.format
|
|
915
|
+
let outputFormat = audio?.output?.format
|
|
916
|
+
let voice = audio?.output?.voice
|
|
917
|
+
|
|
918
|
+
// A `session.update` carrying the transcription field explicitly set to `null` turns
|
|
919
|
+
// transcription off, so track whether the field is *present* rather than whether it is set.
|
|
920
|
+
// Latching the flag on would leave every later turn deferred in `#awaiting` for a transcript
|
|
921
|
+
// that is never coming; ignoring absence is equally required, since a partial update that says
|
|
922
|
+
// nothing about transcription must not disable it.
|
|
923
|
+
if (audio?.input != null && Object.hasOwn(audio.input, 'transcription')) {
|
|
924
|
+
this.#inputTranscriptionEnabled = audio.input.transcription != null
|
|
925
|
+
} else if (Object.hasOwn(session, 'input_audio_transcription')) {
|
|
926
|
+
// Legacy flat field (older SDKs).
|
|
927
|
+
this.#inputTranscriptionEnabled = session.input_audio_transcription != null
|
|
928
|
+
}
|
|
929
|
+
|
|
930
|
+
inputFormat ??= session.input_audio_format
|
|
931
|
+
outputFormat ??= session.output_audio_format
|
|
932
|
+
voice ??= session.voice
|
|
933
|
+
|
|
934
|
+
if (inputFormat != null) {
|
|
935
|
+
this.#inputAudioMime = realtimeAudioFormatToMime(inputFormat)
|
|
936
|
+
this.#sessionConfig.input_audio_format = this.#inputAudioMime
|
|
937
|
+
const rate = toFiniteNumber(inputFormat.rate)
|
|
938
|
+
if (rate) this.#inputAudioRate = rate
|
|
939
|
+
}
|
|
940
|
+
if (outputFormat != null) {
|
|
941
|
+
this.#outputAudioMime = realtimeAudioFormatToMime(outputFormat)
|
|
942
|
+
this.#sessionConfig.output_audio_format = this.#outputAudioMime
|
|
943
|
+
const rate = toFiniteNumber(outputFormat.rate)
|
|
944
|
+
if (rate) this.#outputAudioRate = rate
|
|
945
|
+
}
|
|
946
|
+
if (voice != null) this.#sessionConfig.voice = String(voice)
|
|
947
|
+
}
|
|
948
|
+
|
|
949
|
+
/**
|
|
950
|
+
* @param {Record<string, unknown>} item
|
|
951
|
+
* @param {number} now
|
|
952
|
+
*/
|
|
953
|
+
#absorbInputItem (item, now) {
|
|
954
|
+
if (item == null) return
|
|
955
|
+
|
|
956
|
+
// A tool result the app feeds back becomes a tool result on the next turn's input.
|
|
957
|
+
if (item.type === 'function_call_output') {
|
|
958
|
+
const toolId = String(item.call_id ?? '')
|
|
959
|
+
/** @type {ToolResult} */
|
|
960
|
+
const result = {
|
|
961
|
+
toolId,
|
|
962
|
+
result: item.output == null ? '' : String(item.output),
|
|
963
|
+
type: 'function_call_output',
|
|
964
|
+
}
|
|
965
|
+
// Label the result with the function name carried by the originating call.
|
|
966
|
+
const name = this.#toolCallNames.get(toolId)
|
|
967
|
+
if (name) {
|
|
968
|
+
result.name = name
|
|
969
|
+
this.#toolCallNames.delete(toolId)
|
|
970
|
+
}
|
|
971
|
+
this.#pendingInput.toolResults.push(result)
|
|
972
|
+
return
|
|
973
|
+
}
|
|
974
|
+
|
|
975
|
+
// Only user items contribute to the input turn; skip assistant and system items.
|
|
976
|
+
if (item.role != null && item.role !== 'user') return
|
|
977
|
+
|
|
978
|
+
const { content } = item
|
|
979
|
+
if (!content) return
|
|
980
|
+
|
|
981
|
+
for (const part of content) {
|
|
982
|
+
switch (part?.type) {
|
|
983
|
+
case 'input_text':
|
|
984
|
+
case 'text':
|
|
985
|
+
this.#pendingInput.text += part.text ?? ''
|
|
986
|
+
break
|
|
987
|
+
case 'input_audio':
|
|
988
|
+
case 'audio':
|
|
989
|
+
if (part.audio) {
|
|
990
|
+
this.#pendingInput.audio.append(part.audio, now, this.#inputAudioMime, this.#inputAudioRate)
|
|
991
|
+
}
|
|
992
|
+
if (part.transcript) this.#pendingInput.transcript += part.transcript
|
|
993
|
+
}
|
|
994
|
+
}
|
|
995
|
+
}
|
|
996
|
+
}
|
|
997
|
+
|
|
998
|
+
module.exports = RealtimeSession
|