runlet-harness 0.1.0a1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,4 @@
1
+ * text=auto
2
+ *.py text eol=lf
3
+ *.md text eol=lf
4
+ *.toml text eol=lf
@@ -0,0 +1,31 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+ branches: [main]
8
+
9
+ jobs:
10
+ test:
11
+ name: Python ${{ matrix.python-version }}
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ fail-fast: false
15
+ matrix:
16
+ python-version: ["3.10", "3.11", "3.12"]
17
+
18
+ steps:
19
+ - name: Checkout
20
+ uses: actions/checkout@v4
21
+
22
+ - name: Set up Python
23
+ uses: actions/setup-python@v5
24
+ with:
25
+ python-version: ${{ matrix.python-version }}
26
+
27
+ - name: Install package
28
+ run: python -m pip install -e .
29
+
30
+ - name: Run tests
31
+ run: PYTHONPATH=src python -m unittest discover tests
@@ -0,0 +1,79 @@
1
+ name: Publish
2
+
3
+ on:
4
+ push:
5
+ tags:
6
+ - "v*"
7
+
8
+ jobs:
9
+ build:
10
+ name: Build distribution
11
+ runs-on: ubuntu-latest
12
+
13
+ steps:
14
+ - name: Checkout
15
+ uses: actions/checkout@v4
16
+
17
+ - name: Set up Python
18
+ uses: actions/setup-python@v5
19
+ with:
20
+ python-version: "3.12"
21
+
22
+ - name: Verify tag matches package version
23
+ run: |
24
+ python - <<'PY'
25
+ from pathlib import Path
26
+ import os
27
+ import tomllib
28
+
29
+ ref = os.environ["GITHUB_REF_NAME"]
30
+ tag_version = ref.removeprefix("v")
31
+ data = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))
32
+ package_version = data["project"]["version"]
33
+ if package_version != tag_version:
34
+ raise SystemExit(
35
+ f"Tag version {tag_version!r} does not match project version {package_version!r}"
36
+ )
37
+ PY
38
+
39
+ - name: Install build tools
40
+ run: python -m pip install --upgrade build twine
41
+
42
+ - name: Install package
43
+ run: python -m pip install -e .
44
+
45
+ - name: Run tests
46
+ run: PYTHONPATH=src python -m unittest discover tests
47
+
48
+ - name: Build distributions
49
+ run: python -m build
50
+
51
+ - name: Check distributions
52
+ run: python -m twine check dist/*
53
+
54
+ - name: Upload distributions
55
+ uses: actions/upload-artifact@v4
56
+ with:
57
+ name: python-dist
58
+ path: dist/
59
+
60
+ publish:
61
+ name: Publish to PyPI
62
+ needs: build
63
+ runs-on: ubuntu-latest
64
+ permissions:
65
+ id-token: write
66
+
67
+ environment:
68
+ name: pypi
69
+ url: https://pypi.org/p/runlet-harness
70
+
71
+ steps:
72
+ - name: Download distributions
73
+ uses: actions/download-artifact@v4
74
+ with:
75
+ name: python-dist
76
+ path: dist/
77
+
78
+ - name: Publish distributions
79
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,11 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ .mypy_cache/
7
+ .pyright/
8
+ build/
9
+ dist/
10
+ .venv/
11
+ .env
@@ -0,0 +1,36 @@
1
+ # Repository Instructions
2
+
3
+ These instructions apply to the entire repository.
4
+
5
+ ## Project Identity
6
+
7
+ - Project name: Runlet Harness.
8
+ - Python package name: `runlet_harness`.
9
+ - License: MIT.
10
+ - Purpose: application integration helpers for Runlet.
11
+
12
+ ## Boundary
13
+
14
+ `runlet-harness` may contain third-party platform adapters, environment-based
15
+ configuration helpers, and background delivery lifecycle code. It depends on
16
+ `runlet`.
17
+
18
+ `runlet` must not depend on `runlet-harness`.
19
+
20
+ Do not put core runtime behavior, provider-neutral model contracts, or agent
21
+ execution logic in this package.
22
+
23
+ ## Development Practices
24
+
25
+ - Keep optional SDKs behind optional dependencies and lazy imports.
26
+ - `import runlet_harness` must not import optional SDK packages.
27
+ - Prefer simple Python modules and explicit protocols over framework-heavy
28
+ abstractions.
29
+ - Use fake clients in tests; do not call third-party network services.
30
+ - Keep generated caches such as `__pycache__/` out of commits.
31
+
32
+ Current test command:
33
+
34
+ ```bash
35
+ PYTHONPATH=src python -m unittest discover tests
36
+ ```
@@ -0,0 +1,7 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0a1
4
+
5
+ - Initial package skeleton.
6
+ - Added buffered observability sink support.
7
+ - Added optional Langfuse event sink.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Runlet Harness contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,124 @@
1
+ Metadata-Version: 2.4
2
+ Name: runlet-harness
3
+ Version: 0.1.0a1
4
+ Summary: Application integration helpers for Runlet.
5
+ Project-URL: Homepage, https://github.com/DMIAOCHEN/runlet-harness
6
+ Project-URL: Repository, https://github.com/DMIAOCHEN/runlet-harness
7
+ Project-URL: Issues, https://github.com/DMIAOCHEN/runlet-harness/issues
8
+ Author: Runlet Harness contributors
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: agent,langfuse,observability,runlet,runtime
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.10
21
+ Requires-Dist: runlet>=0.2.0b3
22
+ Provides-Extra: dev
23
+ Requires-Dist: pyright>=1.1.0; extra == 'dev'
24
+ Requires-Dist: ruff>=0.5.0; extra == 'dev'
25
+ Provides-Extra: langfuse
26
+ Requires-Dist: langfuse<5,>=4.7; extra == 'langfuse'
27
+ Description-Content-Type: text/markdown
28
+
29
+ # Runlet Harness
30
+
31
+ [![PyPI version](https://img.shields.io/pypi/v/runlet-harness.svg)](https://pypi.org/project/runlet-harness/)
32
+ [![Python versions](https://img.shields.io/pypi/pyversions/runlet-harness.svg)](https://pypi.org/project/runlet-harness/)
33
+ [![CI](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml/badge.svg)](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml)
34
+
35
+ Application integration helpers for [Runlet](https://github.com/DMIAOCHEN/runlet).
36
+
37
+ `runlet-harness` keeps third-party adapters, environment-based configuration,
38
+ and background delivery concerns outside the small Runlet runtime core.
39
+
40
+ ## Install
41
+
42
+ Base package:
43
+
44
+ ```bash
45
+ pip install runlet-harness
46
+ ```
47
+
48
+ With Langfuse support:
49
+
50
+ ```bash
51
+ pip install "runlet-harness[langfuse]"
52
+ ```
53
+
54
+ ## Langfuse
55
+
56
+ Cloud configuration:
57
+
58
+ ```dotenv
59
+ LANGFUSE_PUBLIC_KEY=pk-lf-...
60
+ LANGFUSE_SECRET_KEY=sk-lf-...
61
+ ```
62
+
63
+ Self-hosted configuration:
64
+
65
+ ```dotenv
66
+ LANGFUSE_PUBLIC_KEY=pk-lf-...
67
+ LANGFUSE_SECRET_KEY=sk-lf-...
68
+ LANGFUSE_BASE_URL=https://langfuse.example.com
69
+ ```
70
+
71
+ Usage:
72
+
73
+ ```python
74
+ from runlet import CompositeEventSink, InMemoryObserver, Runtime
75
+ from runlet_harness.observability import LangfuseEventSink
76
+
77
+ observer = InMemoryObserver()
78
+ langfuse = LangfuseEventSink.from_env()
79
+
80
+ sinks = [observer]
81
+ if langfuse is not None:
82
+ sinks.append(langfuse)
83
+
84
+ runtime = Runtime(event_sink=CompositeEventSink(sinks))
85
+
86
+ try:
87
+ result = await runtime.run(agent, "hello")
88
+ finally:
89
+ if langfuse is not None:
90
+ await langfuse.shutdown()
91
+ ```
92
+
93
+ By default, the Langfuse sink records metadata, status, timing, tool names, and
94
+ final outputs. It does not record inputs, reasoning, streaming deltas, tool
95
+ arguments, tool results, or human-submitted values unless explicitly configured.
96
+ ## Development
97
+
98
+ Run the test suite:
99
+
100
+ ```bash
101
+ PYTHONPATH=src python -m unittest discover tests
102
+ ```
103
+
104
+ Run type checking after installing development dependencies:
105
+
106
+ ```bash
107
+ pyright
108
+ ```
109
+
110
+ ## Release
111
+
112
+ Runlet Harness publishes to PyPI from Git tags through GitHub Actions.
113
+
114
+ Typical release flow:
115
+
116
+ 1. Update the version in `pyproject.toml`.
117
+ 2. Merge to `main`.
118
+ 3. Create a tag such as `v0.1.0a1`.
119
+ 4. Push the tag.
120
+
121
+ ```bash
122
+ git tag v0.1.0a1
123
+ git push origin v0.1.0a1
124
+ ```
@@ -0,0 +1,96 @@
1
+ # Runlet Harness
2
+
3
+ [![PyPI version](https://img.shields.io/pypi/v/runlet-harness.svg)](https://pypi.org/project/runlet-harness/)
4
+ [![Python versions](https://img.shields.io/pypi/pyversions/runlet-harness.svg)](https://pypi.org/project/runlet-harness/)
5
+ [![CI](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml/badge.svg)](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml)
6
+
7
+ Application integration helpers for [Runlet](https://github.com/DMIAOCHEN/runlet).
8
+
9
+ `runlet-harness` keeps third-party adapters, environment-based configuration,
10
+ and background delivery concerns outside the small Runlet runtime core.
11
+
12
+ ## Install
13
+
14
+ Base package:
15
+
16
+ ```bash
17
+ pip install runlet-harness
18
+ ```
19
+
20
+ With Langfuse support:
21
+
22
+ ```bash
23
+ pip install "runlet-harness[langfuse]"
24
+ ```
25
+
26
+ ## Langfuse
27
+
28
+ Cloud configuration:
29
+
30
+ ```dotenv
31
+ LANGFUSE_PUBLIC_KEY=pk-lf-...
32
+ LANGFUSE_SECRET_KEY=sk-lf-...
33
+ ```
34
+
35
+ Self-hosted configuration:
36
+
37
+ ```dotenv
38
+ LANGFUSE_PUBLIC_KEY=pk-lf-...
39
+ LANGFUSE_SECRET_KEY=sk-lf-...
40
+ LANGFUSE_BASE_URL=https://langfuse.example.com
41
+ ```
42
+
43
+ Usage:
44
+
45
+ ```python
46
+ from runlet import CompositeEventSink, InMemoryObserver, Runtime
47
+ from runlet_harness.observability import LangfuseEventSink
48
+
49
+ observer = InMemoryObserver()
50
+ langfuse = LangfuseEventSink.from_env()
51
+
52
+ sinks = [observer]
53
+ if langfuse is not None:
54
+ sinks.append(langfuse)
55
+
56
+ runtime = Runtime(event_sink=CompositeEventSink(sinks))
57
+
58
+ try:
59
+ result = await runtime.run(agent, "hello")
60
+ finally:
61
+ if langfuse is not None:
62
+ await langfuse.shutdown()
63
+ ```
64
+
65
+ By default, the Langfuse sink records metadata, status, timing, tool names, and
66
+ final outputs. It does not record inputs, reasoning, streaming deltas, tool
67
+ arguments, tool results, or human-submitted values unless explicitly configured.
68
+ ## Development
69
+
70
+ Run the test suite:
71
+
72
+ ```bash
73
+ PYTHONPATH=src python -m unittest discover tests
74
+ ```
75
+
76
+ Run type checking after installing development dependencies:
77
+
78
+ ```bash
79
+ pyright
80
+ ```
81
+
82
+ ## Release
83
+
84
+ Runlet Harness publishes to PyPI from Git tags through GitHub Actions.
85
+
86
+ Typical release flow:
87
+
88
+ 1. Update the version in `pyproject.toml`.
89
+ 2. Merge to `main`.
90
+ 3. Create a tag such as `v0.1.0a1`.
91
+ 4. Push the tag.
92
+
93
+ ```bash
94
+ git tag v0.1.0a1
95
+ git push origin v0.1.0a1
96
+ ```
@@ -0,0 +1,57 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "runlet-harness"
7
+ version = "0.1.0a1"
8
+ description = "Application integration helpers for Runlet."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ authors = [
13
+ { name = "Runlet Harness contributors" },
14
+ ]
15
+ keywords = ["agent", "runtime", "observability", "langfuse", "runlet"]
16
+ classifiers = [
17
+ "Development Status :: 3 - Alpha",
18
+ "Intended Audience :: Developers",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Typing :: Typed",
25
+ ]
26
+ dependencies = [
27
+ "runlet>=0.2.0b3",
28
+ ]
29
+
30
+ [project.optional-dependencies]
31
+ langfuse = [
32
+ "langfuse>=4.7,<5",
33
+ ]
34
+ dev = [
35
+ "pyright>=1.1.0",
36
+ "ruff>=0.5.0",
37
+ ]
38
+
39
+ [project.urls]
40
+ Homepage = "https://github.com/DMIAOCHEN/runlet-harness"
41
+ Repository = "https://github.com/DMIAOCHEN/runlet-harness"
42
+ Issues = "https://github.com/DMIAOCHEN/runlet-harness/issues"
43
+
44
+ [tool.hatch.build.targets.wheel]
45
+ packages = ["src/runlet_harness"]
46
+
47
+ [tool.ruff]
48
+ line-length = 100
49
+ target-version = "py310"
50
+
51
+ [tool.ruff.lint]
52
+ select = ["E", "F", "I", "UP", "B", "SIM"]
53
+
54
+ [tool.pyright]
55
+ pythonVersion = "3.10"
56
+ typeCheckingMode = "strict"
57
+ include = ["src", "tests"]
@@ -0,0 +1,5 @@
1
+ """Application integration helpers for Runlet."""
2
+
3
+ __version__ = "0.1.0a1"
4
+
5
+ __all__ = ["__version__"]
@@ -0,0 +1,57 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ from collections.abc import Mapping
5
+
6
+ TRUE_VALUES = {"1", "true", "yes", "y", "on"}
7
+ FALSE_VALUES = {"0", "false", "no", "n", "off"}
8
+
9
+
10
+ def env_str(name: str, *, environ: Mapping[str, str] | None = None) -> str | None:
11
+ source = os.environ if environ is None else environ
12
+ value = source.get(name)
13
+ if value is None:
14
+ return None
15
+ value = value.strip()
16
+ return value or None
17
+
18
+
19
+ def parse_bool(value: str | None, *, name: str) -> bool | None:
20
+ if value is None or value.strip() == "":
21
+ return None
22
+ normalized = value.strip().lower()
23
+ if normalized in TRUE_VALUES:
24
+ return True
25
+ if normalized in FALSE_VALUES:
26
+ return False
27
+ raise ValueError(f"{name} must be a boolean value.")
28
+
29
+
30
+ def env_bool(name: str, *, environ: Mapping[str, str] | None = None) -> bool | None:
31
+ return parse_bool(env_str(name, environ=environ), name=name)
32
+
33
+
34
+ def parse_int(value: str | None, *, name: str) -> int | None:
35
+ if value is None or value.strip() == "":
36
+ return None
37
+ try:
38
+ return int(value)
39
+ except ValueError as error:
40
+ raise ValueError(f"{name} must be an integer.") from error
41
+
42
+
43
+ def env_int(name: str, *, environ: Mapping[str, str] | None = None) -> int | None:
44
+ return parse_int(env_str(name, environ=environ), name=name)
45
+
46
+
47
+ def parse_float(value: str | None, *, name: str) -> float | None:
48
+ if value is None or value.strip() == "":
49
+ return None
50
+ try:
51
+ return float(value)
52
+ except ValueError as error:
53
+ raise ValueError(f"{name} must be a float.") from error
54
+
55
+
56
+ def env_float(name: str, *, environ: Mapping[str, str] | None = None) -> float | None:
57
+ return parse_float(env_str(name, environ=environ), name=name)
@@ -0,0 +1,4 @@
1
+ from runlet_harness.observability.buffering import BufferedEventSink
2
+ from runlet_harness.observability.langfuse import LangfuseConfig, LangfuseEventSink
3
+
4
+ __all__ = ["BufferedEventSink", "LangfuseConfig", "LangfuseEventSink"]
@@ -0,0 +1,80 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ from contextlib import suppress
5
+
6
+ from runlet.core import RuntimeEvent
7
+
8
+
9
+ class BufferedEventSink:
10
+ def __init__(
11
+ self,
12
+ *,
13
+ max_queue_size: int = 1000,
14
+ drop_on_full: bool = True,
15
+ ) -> None:
16
+ if max_queue_size < 1:
17
+ raise ValueError("max_queue_size must be at least 1.")
18
+ self.max_queue_size = max_queue_size
19
+ self.drop_on_full = drop_on_full
20
+ self.dropped_events = 0
21
+ self.processed_events = 0
22
+ self.failed_events = 0
23
+ self._queue: asyncio.Queue[RuntimeEvent] | None = None
24
+ self._worker_task: asyncio.Task[None] | None = None
25
+ self._closed = False
26
+
27
+ async def emit(self, event: RuntimeEvent) -> None:
28
+ if self._closed:
29
+ return
30
+ queue = self._ensure_queue()
31
+ if self.drop_on_full:
32
+ try:
33
+ queue.put_nowait(event)
34
+ except asyncio.QueueFull:
35
+ self.dropped_events += 1
36
+ return
37
+ await queue.put(event)
38
+
39
+ async def flush(self) -> None:
40
+ queue = self._queue
41
+ if queue is not None:
42
+ await queue.join()
43
+ await self._flush_backend()
44
+
45
+ async def shutdown(self) -> None:
46
+ if self._closed:
47
+ return
48
+ self._closed = True
49
+ await self.flush()
50
+ task = self._worker_task
51
+ if task is not None:
52
+ task.cancel()
53
+ with suppress(asyncio.CancelledError):
54
+ await task
55
+ self._worker_task = None
56
+
57
+ def _ensure_queue(self) -> asyncio.Queue[RuntimeEvent]:
58
+ if self._queue is None:
59
+ self._queue = asyncio.Queue(maxsize=self.max_queue_size)
60
+ if self._worker_task is None or self._worker_task.done():
61
+ self._worker_task = asyncio.create_task(self._worker())
62
+ return self._queue
63
+
64
+ async def _worker(self) -> None:
65
+ assert self._queue is not None
66
+ while True:
67
+ event = await self._queue.get()
68
+ try:
69
+ await self._process_event(event)
70
+ self.processed_events += 1
71
+ except Exception:
72
+ self.failed_events += 1
73
+ finally:
74
+ self._queue.task_done()
75
+
76
+ async def _process_event(self, event: RuntimeEvent) -> None:
77
+ raise NotImplementedError
78
+
79
+ async def _flush_backend(self) -> None:
80
+ return None
@@ -0,0 +1,356 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Mapping
4
+ from dataclasses import dataclass, field
5
+ import random
6
+ from typing import Any, Protocol
7
+
8
+ from runlet.core import RuntimeEvent
9
+
10
+ from runlet_harness.config import env_bool, env_float, env_int, env_str
11
+ from runlet_harness.observability.buffering import BufferedEventSink
12
+
13
+ TERMINAL_EVENTS = {"run.completed", "run.interrupted", "policy.stopped"}
14
+
15
+
16
+ def _metadata_map() -> dict[str, Any]:
17
+ return {}
18
+
19
+
20
+ def _bool_env_or(name: str, default: bool, environ: Mapping[str, str] | None) -> bool:
21
+ value = env_bool(name, environ=environ)
22
+ return default if value is None else value
23
+
24
+
25
+ def _float_env_or(name: str, default: float, environ: Mapping[str, str] | None) -> float:
26
+ value = env_float(name, environ=environ)
27
+ return default if value is None else value
28
+
29
+
30
+ def _int_env_or(name: str, default: int, environ: Mapping[str, str] | None) -> int:
31
+ value = env_int(name, environ=environ)
32
+ return default if value is None else value
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class LangfuseConfig:
37
+ public_key: str | None = None
38
+ secret_key: str | None = None
39
+ base_url: str | None = None
40
+ enabled: bool | None = None
41
+ sample_rate: float = 1.0
42
+ max_queue_size: int = 1000
43
+ drop_on_full: bool = True
44
+ capture_input: bool = False
45
+ capture_output: bool = True
46
+ capture_reasoning: bool = False
47
+ capture_stream_deltas: bool = False
48
+ capture_tool_arguments: bool = False
49
+ capture_tool_results: bool = False
50
+ max_payload_chars: int = 4000
51
+ flush_on_terminal_event: bool = False
52
+ debug: bool = False
53
+ metadata: dict[str, Any] = field(default_factory=_metadata_map)
54
+
55
+ @classmethod
56
+ def from_env(cls, environ: Mapping[str, str] | None = None) -> "LangfuseConfig | None":
57
+ enabled = env_bool("RUNLET_LANGFUSE_ENABLED", environ=environ)
58
+ public_key = env_str("LANGFUSE_PUBLIC_KEY", environ=environ)
59
+ secret_key = env_str("LANGFUSE_SECRET_KEY", environ=environ)
60
+ base_url = env_str("LANGFUSE_BASE_URL", environ=environ)
61
+ if enabled is False:
62
+ return None
63
+ if enabled is True and (public_key is None or secret_key is None):
64
+ raise ValueError(
65
+ "RUNLET_LANGFUSE_ENABLED=true requires LANGFUSE_PUBLIC_KEY and "
66
+ "LANGFUSE_SECRET_KEY."
67
+ )
68
+ if enabled is None and (public_key is None or secret_key is None):
69
+ return None
70
+ return cls(
71
+ public_key=public_key,
72
+ secret_key=secret_key,
73
+ base_url=base_url,
74
+ enabled=enabled,
75
+ sample_rate=_float_env_or("RUNLET_LANGFUSE_SAMPLE_RATE", 1.0, environ),
76
+ max_queue_size=_int_env_or("RUNLET_LANGFUSE_MAX_QUEUE_SIZE", 1000, environ),
77
+ drop_on_full=_bool_env_or("RUNLET_LANGFUSE_DROP_ON_FULL", True, environ),
78
+ capture_input=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_INPUT", False, environ),
79
+ capture_output=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_OUTPUT", True, environ),
80
+ capture_reasoning=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_REASONING", False, environ),
81
+ capture_stream_deltas=_bool_env_or(
82
+ "RUNLET_LANGFUSE_CAPTURE_STREAM_DELTAS", False, environ
83
+ ),
84
+ capture_tool_arguments=_bool_env_or(
85
+ "RUNLET_LANGFUSE_CAPTURE_TOOL_ARGUMENTS", False, environ
86
+ ),
87
+ capture_tool_results=_bool_env_or(
88
+ "RUNLET_LANGFUSE_CAPTURE_TOOL_RESULTS", False, environ
89
+ ),
90
+ max_payload_chars=_int_env_or("RUNLET_LANGFUSE_MAX_PAYLOAD_CHARS", 4000, environ),
91
+ flush_on_terminal_event=_bool_env_or(
92
+ "RUNLET_LANGFUSE_FLUSH_ON_TERMINAL_EVENT", False, environ
93
+ ),
94
+ debug=_bool_env_or("RUNLET_LANGFUSE_DEBUG", False, environ),
95
+ )
96
+
97
+
98
+ class LangfuseClientProtocol(Protocol):
99
+ def start_observation(self, **kwargs: Any) -> Any:
100
+ ...
101
+
102
+ def flush(self) -> Any:
103
+ ...
104
+
105
+
106
+ @dataclass
107
+ class _RunState:
108
+ root: Any = None
109
+ model: Any = None
110
+ sampled: bool = True
111
+ tool_stack: list[Any] = field(default_factory=list)
112
+ stream_output: list[str] = field(default_factory=list)
113
+
114
+
115
+ class LangfuseEventSink(BufferedEventSink):
116
+ def __init__(self, config: LangfuseConfig, *, client: Any | None = None) -> None:
117
+ if config.sample_rate < 0 or config.sample_rate > 1:
118
+ raise ValueError("sample_rate must be between 0 and 1.")
119
+ if config.max_payload_chars < 0:
120
+ raise ValueError("max_payload_chars must not be negative.")
121
+ self.config = config
122
+ self.client = client if client is not None else self._create_client(config)
123
+ self._runs: dict[str, _RunState] = {}
124
+ super().__init__(
125
+ max_queue_size=config.max_queue_size,
126
+ drop_on_full=config.drop_on_full,
127
+ )
128
+
129
+ @classmethod
130
+ def from_env(cls, *, client: Any | None = None) -> "LangfuseEventSink | None":
131
+ config = LangfuseConfig.from_env()
132
+ if config is None:
133
+ return None
134
+ return cls(config, client=client)
135
+
136
+ async def _process_event(self, event: RuntimeEvent) -> None:
137
+ state = self._runs.setdefault(event.run_id, _RunState())
138
+ if event.type == "run.started":
139
+ state.sampled = self.config.sample_rate >= 1 or random.random() < self.config.sample_rate
140
+ if not state.sampled:
141
+ return
142
+ state.root = self._start_observation(
143
+ None,
144
+ name="runlet.run",
145
+ as_type="span",
146
+ input=self._payload_value(event, "input") if self.config.capture_input else None,
147
+ metadata=self._metadata(event),
148
+ )
149
+ return
150
+ if not state.sampled:
151
+ if event.type in TERMINAL_EVENTS:
152
+ self._runs.pop(event.run_id, None)
153
+ return
154
+ if event.type == "model.requested":
155
+ state.model = self._start_observation(
156
+ state.root,
157
+ name="runlet.model",
158
+ as_type="generation",
159
+ metadata=self._metadata(event),
160
+ )
161
+ return
162
+ if event.type == "model.completed":
163
+ usage = event.payload.get("usage")
164
+ update: dict[str, Any] = {"metadata": self._metadata(event)}
165
+ if isinstance(usage, int):
166
+ update["usage"] = {"total": usage}
167
+ self._update_observation(state.model, **update)
168
+ self._end_observation(state.model)
169
+ state.model = None
170
+ return
171
+ if event.type == "model.stream.started":
172
+ state.stream_output = []
173
+ state.model = self._start_observation(
174
+ state.root,
175
+ name="runlet.model.stream",
176
+ as_type="generation",
177
+ metadata=self._metadata(event),
178
+ )
179
+ return
180
+ if event.type == "model.stream.delta":
181
+ if self.config.capture_stream_deltas:
182
+ delta = self._payload_value(event, "delta")
183
+ if isinstance(delta, str):
184
+ state.stream_output.append(delta)
185
+ return
186
+ if event.type == "model.stream.completed":
187
+ output = None
188
+ if self.config.capture_stream_deltas:
189
+ output = self._truncate("".join(state.stream_output))
190
+ self._update_observation(state.model, output=output, metadata=self._metadata(event))
191
+ self._end_observation(state.model)
192
+ state.model = None
193
+ state.stream_output = []
194
+ return
195
+ if event.type == "tool.started":
196
+ tool_name = str(event.payload.get("name", "unknown"))
197
+ observation = self._start_observation(
198
+ state.root,
199
+ name=f"runlet.tool.{tool_name}",
200
+ as_type="span",
201
+ metadata=self._metadata(event),
202
+ )
203
+ state.tool_stack.append(observation)
204
+ return
205
+ if event.type == "tool.completed":
206
+ observation = state.tool_stack.pop() if state.tool_stack else None
207
+ self._update_observation(observation, metadata=self._metadata(event))
208
+ self._end_observation(observation)
209
+ return
210
+ if event.type in {"human.requested", "human.responded", "human.response_rejected", "run.resumed"}:
211
+ observation = self._start_observation(
212
+ state.root,
213
+ name=f"runlet.{event.type}",
214
+ as_type="event",
215
+ metadata=self._metadata(event),
216
+ )
217
+ self._end_observation(observation)
218
+ return
219
+ if event.type in TERMINAL_EVENTS:
220
+ update = {"metadata": self._metadata(event)}
221
+ output = self._terminal_output(event)
222
+ if output is not None:
223
+ update["output"] = output
224
+ self._update_observation(state.root, **update)
225
+ self._end_observation(state.root)
226
+ self._runs.pop(event.run_id, None)
227
+ if self.config.flush_on_terminal_event:
228
+ await self._flush_backend()
229
+
230
+ async def _flush_backend(self) -> None:
231
+ result = self.client.flush()
232
+ if hasattr(result, "__await__"):
233
+ await result
234
+
235
+ @staticmethod
236
+ def _create_client(config: LangfuseConfig) -> Any:
237
+ if config.public_key is None or config.secret_key is None:
238
+ raise ValueError("Langfuse public_key and secret_key are required when creating a client.")
239
+ try:
240
+ from langfuse import Langfuse # type: ignore[import-not-found]
241
+ except ImportError as error:
242
+ raise RuntimeError(
243
+ "Langfuse support requires the optional dependency. Install it with "
244
+ "pip install \"runlet-harness[langfuse]\"."
245
+ ) from error
246
+ kwargs: dict[str, Any] = {
247
+ "public_key": config.public_key,
248
+ "secret_key": config.secret_key,
249
+ }
250
+ if config.base_url is not None:
251
+ kwargs["base_url"] = config.base_url
252
+ try:
253
+ return Langfuse(**kwargs)
254
+ except TypeError:
255
+ if "base_url" in kwargs:
256
+ kwargs["host"] = kwargs.pop("base_url")
257
+ return Langfuse(**kwargs)
258
+
259
+ def _metadata(self, event: RuntimeEvent) -> dict[str, Any]:
260
+ metadata: dict[str, Any] = {
261
+ "runlet.event_id": event.id,
262
+ "runlet.event_type": event.type,
263
+ "runlet.run_id": event.run_id,
264
+ "runlet.timestamp": event.timestamp.isoformat(),
265
+ "runlet.severity": event.severity,
266
+ **self.config.metadata,
267
+ }
268
+ if event.agent_name is not None:
269
+ metadata["runlet.agent_name"] = event.agent_name
270
+ if event.step_id is not None:
271
+ metadata["runlet.step_id"] = event.step_id
272
+ if event.span_id is not None:
273
+ metadata["runlet.span_id"] = event.span_id
274
+ if event.parent_span_id is not None:
275
+ metadata["runlet.parent_span_id"] = event.parent_span_id
276
+ for key, value in event.attributes.items():
277
+ metadata[f"runlet.attribute.{key}"] = self._sanitize_value(value)
278
+ for key, value in self._safe_payload(event).items():
279
+ metadata[f"runlet.payload.{key}"] = value
280
+ return metadata
281
+
282
+ def _safe_payload(self, event: RuntimeEvent) -> dict[str, Any]:
283
+ payload: dict[str, Any] = {}
284
+ for key, value in event.payload.items():
285
+ if key == "input" and not self.config.capture_input:
286
+ continue
287
+ if key == "output" and not self.config.capture_output:
288
+ continue
289
+ if key == "reasoning" and not self.config.capture_reasoning:
290
+ continue
291
+ if key == "delta" and not self.config.capture_stream_deltas:
292
+ continue
293
+ if key in {"arguments", "tool_arguments"} and not self.config.capture_tool_arguments:
294
+ continue
295
+ if key in {"result", "tool_result"} and not self.config.capture_tool_results:
296
+ continue
297
+ payload[key] = self._sanitize_value(value)
298
+ return payload
299
+
300
+ def _terminal_output(self, event: RuntimeEvent) -> Any | None:
301
+ if not self.config.capture_output:
302
+ return None
303
+ output = self._payload_value(event, "output")
304
+ if output is None:
305
+ return None
306
+ return self._sanitize_value(output)
307
+
308
+ def _payload_value(self, event: RuntimeEvent, key: str) -> Any | None:
309
+ if key not in event.payload:
310
+ return None
311
+ return self._sanitize_value(event.payload[key])
312
+
313
+ def _sanitize_value(self, value: Any) -> Any:
314
+ if isinstance(value, str):
315
+ return self._truncate(value)
316
+ if isinstance(value, (int, float, bool)) or value is None:
317
+ return value
318
+ if isinstance(value, Mapping):
319
+ return {str(key): self._sanitize_value(item) for key, item in value.items()}
320
+ if isinstance(value, (list, tuple)):
321
+ return [self._sanitize_value(item) for item in value]
322
+ return self._truncate(repr(value))
323
+
324
+ def _truncate(self, value: str) -> str:
325
+ limit = self.config.max_payload_chars
326
+ if limit == 0:
327
+ return ""
328
+ if len(value) <= limit:
329
+ return value
330
+ return value[:limit]
331
+
332
+ def _start_observation(self, parent: Any, **kwargs: Any) -> Any:
333
+ target = parent if parent is not None and hasattr(parent, "start_observation") else self.client
334
+ starter = getattr(target, "start_observation")
335
+ kwargs = {key: value for key, value in kwargs.items() if value is not None}
336
+ try:
337
+ return starter(**kwargs)
338
+ except TypeError:
339
+ fallback = dict(kwargs)
340
+ if "as_type" in fallback:
341
+ fallback["type"] = fallback.pop("as_type")
342
+ return starter(**fallback)
343
+
344
+ @staticmethod
345
+ def _update_observation(observation: Any, **kwargs: Any) -> None:
346
+ if observation is None or not hasattr(observation, "update"):
347
+ return
348
+ clean_kwargs = {key: value for key, value in kwargs.items() if value is not None}
349
+ if clean_kwargs:
350
+ observation.update(**clean_kwargs)
351
+
352
+ @staticmethod
353
+ def _end_observation(observation: Any) -> None:
354
+ if observation is None or not hasattr(observation, "end"):
355
+ return
356
+ observation.end()
File without changes
@@ -0,0 +1,65 @@
1
+ import asyncio
2
+ import unittest
3
+
4
+ from runlet.core import RuntimeEvent
5
+ from runlet_harness.observability.buffering import BufferedEventSink
6
+
7
+
8
+ class RecordingSink(BufferedEventSink):
9
+ def __init__(self, **kwargs):
10
+ super().__init__(**kwargs)
11
+ self.events = []
12
+ self.flushed = 0
13
+ self.fail = False
14
+
15
+ async def _process_event(self, event: RuntimeEvent) -> None:
16
+ if self.fail:
17
+ raise RuntimeError("boom")
18
+ self.events.append(event)
19
+
20
+ async def _flush_backend(self) -> None:
21
+ self.flushed += 1
22
+
23
+
24
+ class BufferedEventSinkTests(unittest.IsolatedAsyncioTestCase):
25
+ async def test_emit_processes_events(self) -> None:
26
+ sink = RecordingSink()
27
+ event = RuntimeEvent(type="run.started", run_id="run_1")
28
+
29
+ await sink.emit(event)
30
+ await sink.flush()
31
+
32
+ self.assertEqual(sink.events, [event])
33
+ self.assertEqual(sink.processed_events, 1)
34
+ await sink.shutdown()
35
+
36
+ async def test_queue_full_drops_when_configured(self) -> None:
37
+ sink = RecordingSink(max_queue_size=1, drop_on_full=True)
38
+ await sink.emit(RuntimeEvent(type="one", run_id="run_1"))
39
+ await sink.emit(RuntimeEvent(type="two", run_id="run_1"))
40
+
41
+ self.assertGreaterEqual(sink.dropped_events, 0)
42
+ await sink.shutdown()
43
+
44
+ async def test_worker_errors_do_not_propagate(self) -> None:
45
+ sink = RecordingSink()
46
+ sink.fail = True
47
+
48
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
49
+ await sink.flush()
50
+
51
+ self.assertEqual(sink.failed_events, 1)
52
+ await sink.shutdown()
53
+
54
+ async def test_shutdown_is_idempotent(self) -> None:
55
+ sink = RecordingSink()
56
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
57
+ await sink.shutdown()
58
+ await sink.shutdown()
59
+ self.assertEqual(sink.flushed, 1)
60
+
61
+ async def test_non_dropping_queue_waits_for_space(self) -> None:
62
+ sink = RecordingSink(max_queue_size=1, drop_on_full=False)
63
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
64
+ await asyncio.wait_for(sink.emit(RuntimeEvent(type="run.completed", run_id="run_1")), 1)
65
+ await sink.shutdown()
@@ -0,0 +1,31 @@
1
+ import unittest
2
+
3
+ from runlet_harness.config import env_bool, env_float, env_int, env_str, parse_bool
4
+
5
+
6
+ class ConfigTests(unittest.TestCase):
7
+ def test_env_str_treats_blank_as_missing(self) -> None:
8
+ self.assertIsNone(env_str("MISSING", environ={}))
9
+ self.assertIsNone(env_str("EMPTY", environ={"EMPTY": " "}))
10
+ self.assertEqual(env_str("VALUE", environ={"VALUE": " hello "}), "hello")
11
+
12
+ def test_bool_parsing(self) -> None:
13
+ for value in ("true", "1", "yes", "on"):
14
+ self.assertIs(parse_bool(value, name="FLAG"), True)
15
+ for value in ("false", "0", "no", "off"):
16
+ self.assertIs(parse_bool(value, name="FLAG"), False)
17
+ with self.assertRaisesRegex(ValueError, "FLAG"):
18
+ parse_bool("maybe", name="FLAG")
19
+
20
+ def test_number_parsing(self) -> None:
21
+ self.assertEqual(env_int("COUNT", environ={"COUNT": "5"}), 5)
22
+ self.assertEqual(env_float("RATE", environ={"RATE": "0.25"}), 0.25)
23
+ with self.assertRaisesRegex(ValueError, "COUNT"):
24
+ env_int("COUNT", environ={"COUNT": "x"})
25
+ with self.assertRaisesRegex(ValueError, "RATE"):
26
+ env_float("RATE", environ={"RATE": "x"})
27
+
28
+ def test_env_bool(self) -> None:
29
+ self.assertIs(env_bool("FLAG", environ={"FLAG": "yes"}), True)
30
+ self.assertIs(env_bool("FLAG", environ={"FLAG": "no"}), False)
31
+ self.assertIsNone(env_bool("FLAG", environ={}))
@@ -0,0 +1,189 @@
1
+ import os
2
+ import unittest
3
+ from unittest.mock import patch
4
+
5
+ from runlet import Agent, Message, Runtime
6
+ from runlet.core import RuntimeEvent
7
+ from runlet.core.models import ModelResponse
8
+ from runlet.testing import FakeModelProvider
9
+ from runlet_harness.observability.langfuse import LangfuseConfig, LangfuseEventSink
10
+
11
+
12
+ class FakeObservation:
13
+ def __init__(self, client, **kwargs):
14
+ self.client = client
15
+ self.kwargs = kwargs
16
+ self.updates = []
17
+ self.ended = False
18
+ self.children = []
19
+
20
+ def start_observation(self, **kwargs):
21
+ child = FakeObservation(self.client, **kwargs)
22
+ self.children.append(child)
23
+ self.client.observations.append(child)
24
+ return child
25
+
26
+ def update(self, **kwargs):
27
+ self.updates.append(kwargs)
28
+
29
+ def end(self):
30
+ self.ended = True
31
+
32
+
33
+ class FakeClient:
34
+ def __init__(self):
35
+ self.observations = []
36
+ self.flushed = 0
37
+ self.fail = False
38
+
39
+ def start_observation(self, **kwargs):
40
+ if self.fail:
41
+ raise RuntimeError("langfuse failed")
42
+ observation = FakeObservation(self, **kwargs)
43
+ self.observations.append(observation)
44
+ return observation
45
+
46
+ def flush(self):
47
+ self.flushed += 1
48
+
49
+
50
+ class LangfuseConfigTests(unittest.TestCase):
51
+ def test_from_env_returns_none_without_keys(self) -> None:
52
+ with patch.dict(os.environ, {}, clear=True):
53
+ self.assertIsNone(LangfuseEventSink.from_env(client=FakeClient()))
54
+
55
+ def test_from_env_can_be_disabled(self) -> None:
56
+ env = {
57
+ "RUNLET_LANGFUSE_ENABLED": "false",
58
+ "LANGFUSE_PUBLIC_KEY": "pk",
59
+ "LANGFUSE_SECRET_KEY": "sk",
60
+ }
61
+ with patch.dict(os.environ, env, clear=True):
62
+ self.assertIsNone(LangfuseEventSink.from_env(client=FakeClient()))
63
+
64
+ def test_forced_enable_requires_keys(self) -> None:
65
+ with patch.dict(os.environ, {"RUNLET_LANGFUSE_ENABLED": "true"}, clear=True):
66
+ with self.assertRaisesRegex(ValueError, "LANGFUSE_PUBLIC_KEY"):
67
+ LangfuseEventSink.from_env(client=FakeClient())
68
+
69
+ def test_env_base_url_is_loaded(self) -> None:
70
+ env = {
71
+ "LANGFUSE_PUBLIC_KEY": "pk",
72
+ "LANGFUSE_SECRET_KEY": "sk",
73
+ "LANGFUSE_BASE_URL": "https://langfuse.example.com",
74
+ }
75
+ with patch.dict(os.environ, env, clear=True):
76
+ config = LangfuseConfig.from_env()
77
+ assert config is not None
78
+ self.assertEqual(config.base_url, "https://langfuse.example.com")
79
+
80
+ def test_env_explicit_zero_values_are_preserved(self) -> None:
81
+ env = {
82
+ "LANGFUSE_PUBLIC_KEY": "pk",
83
+ "LANGFUSE_SECRET_KEY": "sk",
84
+ "RUNLET_LANGFUSE_SAMPLE_RATE": "0",
85
+ "RUNLET_LANGFUSE_MAX_PAYLOAD_CHARS": "0",
86
+ }
87
+ with patch.dict(os.environ, env, clear=True):
88
+ config = LangfuseConfig.from_env()
89
+ assert config is not None
90
+ self.assertEqual(config.sample_rate, 0)
91
+ self.assertEqual(config.max_payload_chars, 0)
92
+
93
+ def test_missing_sdk_error_is_clear(self) -> None:
94
+ with patch("runlet_harness.observability.langfuse.LangfuseEventSink._create_client") as create:
95
+ create.side_effect = RuntimeError('Install it with pip install "runlet-harness[langfuse]".')
96
+ with self.assertRaisesRegex(RuntimeError, "runlet-harness\\[langfuse\\]"):
97
+ LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"))
98
+
99
+
100
+ class LangfuseEventSinkTests(unittest.IsolatedAsyncioTestCase):
101
+ async def test_maps_run_model_tool_and_terminal_events(self) -> None:
102
+ client = FakeClient()
103
+ sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
104
+
105
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1", agent_name="agent", payload={"input": "secret"}))
106
+ await sink.emit(RuntimeEvent(type="model.requested", run_id="run_1", agent_name="agent"))
107
+ await sink.emit(RuntimeEvent(type="model.completed", run_id="run_1", agent_name="agent", payload={"usage": 7}))
108
+ await sink.emit(RuntimeEvent(type="tool.started", run_id="run_1", agent_name="agent", payload={"name": "lookup"}))
109
+ await sink.emit(RuntimeEvent(type="tool.completed", run_id="run_1", agent_name="agent", payload={"name": "lookup"}))
110
+ await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1", agent_name="agent", payload={"output": "done"}))
111
+ await sink.shutdown()
112
+
113
+ names = [observation.kwargs["name"] for observation in client.observations]
114
+ self.assertIn("runlet.run", names)
115
+ self.assertIn("runlet.model", names)
116
+ self.assertIn("runlet.tool.lookup", names)
117
+ root = client.observations[0]
118
+ self.assertNotIn("input", root.kwargs)
119
+ self.assertTrue(root.ended)
120
+ self.assertEqual(root.updates[-1]["output"], "done")
121
+
122
+ async def test_stream_delta_is_ignored_by_default(self) -> None:
123
+ client = FakeClient()
124
+ sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
125
+
126
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
127
+ await sink.emit(RuntimeEvent(type="model.stream.started", run_id="run_1"))
128
+ await sink.emit(RuntimeEvent(type="model.stream.delta", run_id="run_1", payload={"delta": "private"}))
129
+ await sink.emit(RuntimeEvent(type="model.stream.completed", run_id="run_1"))
130
+ await sink.shutdown()
131
+
132
+ stream = [observation for observation in client.observations if observation.kwargs["name"] == "runlet.model.stream"][0]
133
+ self.assertNotIn("output", stream.updates[-1])
134
+
135
+ async def test_stream_delta_can_be_captured_and_truncated(self) -> None:
136
+ client = FakeClient()
137
+ config = LangfuseConfig(
138
+ public_key="pk",
139
+ secret_key="sk",
140
+ capture_stream_deltas=True,
141
+ max_payload_chars=4,
142
+ )
143
+ sink = LangfuseEventSink(config, client=client)
144
+
145
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
146
+ await sink.emit(RuntimeEvent(type="model.stream.started", run_id="run_1"))
147
+ await sink.emit(RuntimeEvent(type="model.stream.delta", run_id="run_1", payload={"delta": "private"}))
148
+ await sink.emit(RuntimeEvent(type="model.stream.completed", run_id="run_1"))
149
+ await sink.shutdown()
150
+
151
+ stream = [observation for observation in client.observations if observation.kwargs["name"] == "runlet.model.stream"][0]
152
+ self.assertEqual(stream.updates[-1]["output"], "priv")
153
+
154
+ async def test_client_errors_do_not_fail_runtime(self) -> None:
155
+ client = FakeClient()
156
+ client.fail = True
157
+ sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
158
+ model = FakeModelProvider([ModelResponse(message=Message.assistant("ok"))])
159
+ agent = Agent(name="assistant", instructions="Help.", model=model)
160
+
161
+ result = await Runtime(event_sink=sink).run(agent, "hi")
162
+ await sink.shutdown()
163
+
164
+ self.assertEqual(result.output, "ok")
165
+ self.assertGreater(sink.failed_events, 0)
166
+
167
+ async def test_sample_rate_zero_skips_observations(self) -> None:
168
+ client = FakeClient()
169
+ sink = LangfuseEventSink(
170
+ LangfuseConfig(public_key="pk", secret_key="sk", sample_rate=0),
171
+ client=client,
172
+ )
173
+
174
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
175
+ await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1"))
176
+ await sink.shutdown()
177
+
178
+ self.assertEqual(client.observations, [])
179
+
180
+ async def test_flush_on_terminal_event(self) -> None:
181
+ client = FakeClient()
182
+ config = LangfuseConfig(public_key="pk", secret_key="sk", flush_on_terminal_event=True)
183
+ sink = LangfuseEventSink(config, client=client)
184
+
185
+ await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
186
+ await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1"))
187
+ await sink.shutdown()
188
+
189
+ self.assertGreaterEqual(client.flushed, 1)