runlet-harness 0.1.0a1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- runlet_harness-0.1.0a1/.gitattributes +4 -0
- runlet_harness-0.1.0a1/.github/workflows/ci.yml +31 -0
- runlet_harness-0.1.0a1/.github/workflows/publish.yml +79 -0
- runlet_harness-0.1.0a1/.gitignore +11 -0
- runlet_harness-0.1.0a1/AGENTS.md +36 -0
- runlet_harness-0.1.0a1/CHANGELOG.md +7 -0
- runlet_harness-0.1.0a1/LICENSE +21 -0
- runlet_harness-0.1.0a1/PKG-INFO +124 -0
- runlet_harness-0.1.0a1/README.md +96 -0
- runlet_harness-0.1.0a1/pyproject.toml +57 -0
- runlet_harness-0.1.0a1/src/runlet_harness/__init__.py +5 -0
- runlet_harness-0.1.0a1/src/runlet_harness/config.py +57 -0
- runlet_harness-0.1.0a1/src/runlet_harness/observability/__init__.py +4 -0
- runlet_harness-0.1.0a1/src/runlet_harness/observability/buffering.py +80 -0
- runlet_harness-0.1.0a1/src/runlet_harness/observability/langfuse.py +356 -0
- runlet_harness-0.1.0a1/src/runlet_harness/py.typed +0 -0
- runlet_harness-0.1.0a1/tests/test_buffering.py +65 -0
- runlet_harness-0.1.0a1/tests/test_config.py +31 -0
- runlet_harness-0.1.0a1/tests/test_langfuse.py +189 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
test:
|
|
11
|
+
name: Python ${{ matrix.python-version }}
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
strategy:
|
|
14
|
+
fail-fast: false
|
|
15
|
+
matrix:
|
|
16
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
17
|
+
|
|
18
|
+
steps:
|
|
19
|
+
- name: Checkout
|
|
20
|
+
uses: actions/checkout@v4
|
|
21
|
+
|
|
22
|
+
- name: Set up Python
|
|
23
|
+
uses: actions/setup-python@v5
|
|
24
|
+
with:
|
|
25
|
+
python-version: ${{ matrix.python-version }}
|
|
26
|
+
|
|
27
|
+
- name: Install package
|
|
28
|
+
run: python -m pip install -e .
|
|
29
|
+
|
|
30
|
+
- name: Run tests
|
|
31
|
+
run: PYTHONPATH=src python -m unittest discover tests
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
name: Publish
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
build:
|
|
10
|
+
name: Build distribution
|
|
11
|
+
runs-on: ubuntu-latest
|
|
12
|
+
|
|
13
|
+
steps:
|
|
14
|
+
- name: Checkout
|
|
15
|
+
uses: actions/checkout@v4
|
|
16
|
+
|
|
17
|
+
- name: Set up Python
|
|
18
|
+
uses: actions/setup-python@v5
|
|
19
|
+
with:
|
|
20
|
+
python-version: "3.12"
|
|
21
|
+
|
|
22
|
+
- name: Verify tag matches package version
|
|
23
|
+
run: |
|
|
24
|
+
python - <<'PY'
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
import os
|
|
27
|
+
import tomllib
|
|
28
|
+
|
|
29
|
+
ref = os.environ["GITHUB_REF_NAME"]
|
|
30
|
+
tag_version = ref.removeprefix("v")
|
|
31
|
+
data = tomllib.loads(Path("pyproject.toml").read_text(encoding="utf-8"))
|
|
32
|
+
package_version = data["project"]["version"]
|
|
33
|
+
if package_version != tag_version:
|
|
34
|
+
raise SystemExit(
|
|
35
|
+
f"Tag version {tag_version!r} does not match project version {package_version!r}"
|
|
36
|
+
)
|
|
37
|
+
PY
|
|
38
|
+
|
|
39
|
+
- name: Install build tools
|
|
40
|
+
run: python -m pip install --upgrade build twine
|
|
41
|
+
|
|
42
|
+
- name: Install package
|
|
43
|
+
run: python -m pip install -e .
|
|
44
|
+
|
|
45
|
+
- name: Run tests
|
|
46
|
+
run: PYTHONPATH=src python -m unittest discover tests
|
|
47
|
+
|
|
48
|
+
- name: Build distributions
|
|
49
|
+
run: python -m build
|
|
50
|
+
|
|
51
|
+
- name: Check distributions
|
|
52
|
+
run: python -m twine check dist/*
|
|
53
|
+
|
|
54
|
+
- name: Upload distributions
|
|
55
|
+
uses: actions/upload-artifact@v4
|
|
56
|
+
with:
|
|
57
|
+
name: python-dist
|
|
58
|
+
path: dist/
|
|
59
|
+
|
|
60
|
+
publish:
|
|
61
|
+
name: Publish to PyPI
|
|
62
|
+
needs: build
|
|
63
|
+
runs-on: ubuntu-latest
|
|
64
|
+
permissions:
|
|
65
|
+
id-token: write
|
|
66
|
+
|
|
67
|
+
environment:
|
|
68
|
+
name: pypi
|
|
69
|
+
url: https://pypi.org/p/runlet-harness
|
|
70
|
+
|
|
71
|
+
steps:
|
|
72
|
+
- name: Download distributions
|
|
73
|
+
uses: actions/download-artifact@v4
|
|
74
|
+
with:
|
|
75
|
+
name: python-dist
|
|
76
|
+
path: dist/
|
|
77
|
+
|
|
78
|
+
- name: Publish distributions
|
|
79
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Repository Instructions
|
|
2
|
+
|
|
3
|
+
These instructions apply to the entire repository.
|
|
4
|
+
|
|
5
|
+
## Project Identity
|
|
6
|
+
|
|
7
|
+
- Project name: Runlet Harness.
|
|
8
|
+
- Python package name: `runlet_harness`.
|
|
9
|
+
- License: MIT.
|
|
10
|
+
- Purpose: application integration helpers for Runlet.
|
|
11
|
+
|
|
12
|
+
## Boundary
|
|
13
|
+
|
|
14
|
+
`runlet-harness` may contain third-party platform adapters, environment-based
|
|
15
|
+
configuration helpers, and background delivery lifecycle code. It depends on
|
|
16
|
+
`runlet`.
|
|
17
|
+
|
|
18
|
+
`runlet` must not depend on `runlet-harness`.
|
|
19
|
+
|
|
20
|
+
Do not put core runtime behavior, provider-neutral model contracts, or agent
|
|
21
|
+
execution logic in this package.
|
|
22
|
+
|
|
23
|
+
## Development Practices
|
|
24
|
+
|
|
25
|
+
- Keep optional SDKs behind optional dependencies and lazy imports.
|
|
26
|
+
- `import runlet_harness` must not import optional SDK packages.
|
|
27
|
+
- Prefer simple Python modules and explicit protocols over framework-heavy
|
|
28
|
+
abstractions.
|
|
29
|
+
- Use fake clients in tests; do not call third-party network services.
|
|
30
|
+
- Keep generated caches such as `__pycache__/` out of commits.
|
|
31
|
+
|
|
32
|
+
Current test command:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
PYTHONPATH=src python -m unittest discover tests
|
|
36
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Runlet Harness contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: runlet-harness
|
|
3
|
+
Version: 0.1.0a1
|
|
4
|
+
Summary: Application integration helpers for Runlet.
|
|
5
|
+
Project-URL: Homepage, https://github.com/DMIAOCHEN/runlet-harness
|
|
6
|
+
Project-URL: Repository, https://github.com/DMIAOCHEN/runlet-harness
|
|
7
|
+
Project-URL: Issues, https://github.com/DMIAOCHEN/runlet-harness/issues
|
|
8
|
+
Author: Runlet Harness contributors
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: agent,langfuse,observability,runlet,runtime
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Requires-Dist: runlet>=0.2.0b3
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Requires-Dist: pyright>=1.1.0; extra == 'dev'
|
|
24
|
+
Requires-Dist: ruff>=0.5.0; extra == 'dev'
|
|
25
|
+
Provides-Extra: langfuse
|
|
26
|
+
Requires-Dist: langfuse<5,>=4.7; extra == 'langfuse'
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# Runlet Harness
|
|
30
|
+
|
|
31
|
+
[](https://pypi.org/project/runlet-harness/)
|
|
32
|
+
[](https://pypi.org/project/runlet-harness/)
|
|
33
|
+
[](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml)
|
|
34
|
+
|
|
35
|
+
Application integration helpers for [Runlet](https://github.com/DMIAOCHEN/runlet).
|
|
36
|
+
|
|
37
|
+
`runlet-harness` keeps third-party adapters, environment-based configuration,
|
|
38
|
+
and background delivery concerns outside the small Runlet runtime core.
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
Base package:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install runlet-harness
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
With Langfuse support:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install "runlet-harness[langfuse]"
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Langfuse
|
|
55
|
+
|
|
56
|
+
Cloud configuration:
|
|
57
|
+
|
|
58
|
+
```dotenv
|
|
59
|
+
LANGFUSE_PUBLIC_KEY=pk-lf-...
|
|
60
|
+
LANGFUSE_SECRET_KEY=sk-lf-...
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Self-hosted configuration:
|
|
64
|
+
|
|
65
|
+
```dotenv
|
|
66
|
+
LANGFUSE_PUBLIC_KEY=pk-lf-...
|
|
67
|
+
LANGFUSE_SECRET_KEY=sk-lf-...
|
|
68
|
+
LANGFUSE_BASE_URL=https://langfuse.example.com
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Usage:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from runlet import CompositeEventSink, InMemoryObserver, Runtime
|
|
75
|
+
from runlet_harness.observability import LangfuseEventSink
|
|
76
|
+
|
|
77
|
+
observer = InMemoryObserver()
|
|
78
|
+
langfuse = LangfuseEventSink.from_env()
|
|
79
|
+
|
|
80
|
+
sinks = [observer]
|
|
81
|
+
if langfuse is not None:
|
|
82
|
+
sinks.append(langfuse)
|
|
83
|
+
|
|
84
|
+
runtime = Runtime(event_sink=CompositeEventSink(sinks))
|
|
85
|
+
|
|
86
|
+
try:
|
|
87
|
+
result = await runtime.run(agent, "hello")
|
|
88
|
+
finally:
|
|
89
|
+
if langfuse is not None:
|
|
90
|
+
await langfuse.shutdown()
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
By default, the Langfuse sink records metadata, status, timing, tool names, and
|
|
94
|
+
final outputs. It does not record inputs, reasoning, streaming deltas, tool
|
|
95
|
+
arguments, tool results, or human-submitted values unless explicitly configured.
|
|
96
|
+
## Development
|
|
97
|
+
|
|
98
|
+
Run the test suite:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
PYTHONPATH=src python -m unittest discover tests
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Run type checking after installing development dependencies:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
pyright
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Release
|
|
111
|
+
|
|
112
|
+
Runlet Harness publishes to PyPI from Git tags through GitHub Actions.
|
|
113
|
+
|
|
114
|
+
Typical release flow:
|
|
115
|
+
|
|
116
|
+
1. Update the version in `pyproject.toml`.
|
|
117
|
+
2. Merge to `main`.
|
|
118
|
+
3. Create a tag such as `v0.1.0a1`.
|
|
119
|
+
4. Push the tag.
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
git tag v0.1.0a1
|
|
123
|
+
git push origin v0.1.0a1
|
|
124
|
+
```
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Runlet Harness
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/runlet-harness/)
|
|
4
|
+
[](https://pypi.org/project/runlet-harness/)
|
|
5
|
+
[](https://github.com/DMIAOCHEN/runlet-harness/actions/workflows/ci.yml)
|
|
6
|
+
|
|
7
|
+
Application integration helpers for [Runlet](https://github.com/DMIAOCHEN/runlet).
|
|
8
|
+
|
|
9
|
+
`runlet-harness` keeps third-party adapters, environment-based configuration,
|
|
10
|
+
and background delivery concerns outside the small Runlet runtime core.
|
|
11
|
+
|
|
12
|
+
## Install
|
|
13
|
+
|
|
14
|
+
Base package:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install runlet-harness
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
With Langfuse support:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install "runlet-harness[langfuse]"
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Langfuse
|
|
27
|
+
|
|
28
|
+
Cloud configuration:
|
|
29
|
+
|
|
30
|
+
```dotenv
|
|
31
|
+
LANGFUSE_PUBLIC_KEY=pk-lf-...
|
|
32
|
+
LANGFUSE_SECRET_KEY=sk-lf-...
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Self-hosted configuration:
|
|
36
|
+
|
|
37
|
+
```dotenv
|
|
38
|
+
LANGFUSE_PUBLIC_KEY=pk-lf-...
|
|
39
|
+
LANGFUSE_SECRET_KEY=sk-lf-...
|
|
40
|
+
LANGFUSE_BASE_URL=https://langfuse.example.com
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Usage:
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from runlet import CompositeEventSink, InMemoryObserver, Runtime
|
|
47
|
+
from runlet_harness.observability import LangfuseEventSink
|
|
48
|
+
|
|
49
|
+
observer = InMemoryObserver()
|
|
50
|
+
langfuse = LangfuseEventSink.from_env()
|
|
51
|
+
|
|
52
|
+
sinks = [observer]
|
|
53
|
+
if langfuse is not None:
|
|
54
|
+
sinks.append(langfuse)
|
|
55
|
+
|
|
56
|
+
runtime = Runtime(event_sink=CompositeEventSink(sinks))
|
|
57
|
+
|
|
58
|
+
try:
|
|
59
|
+
result = await runtime.run(agent, "hello")
|
|
60
|
+
finally:
|
|
61
|
+
if langfuse is not None:
|
|
62
|
+
await langfuse.shutdown()
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
By default, the Langfuse sink records metadata, status, timing, tool names, and
|
|
66
|
+
final outputs. It does not record inputs, reasoning, streaming deltas, tool
|
|
67
|
+
arguments, tool results, or human-submitted values unless explicitly configured.
|
|
68
|
+
## Development
|
|
69
|
+
|
|
70
|
+
Run the test suite:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
PYTHONPATH=src python -m unittest discover tests
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Run type checking after installing development dependencies:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
pyright
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Release
|
|
83
|
+
|
|
84
|
+
Runlet Harness publishes to PyPI from Git tags through GitHub Actions.
|
|
85
|
+
|
|
86
|
+
Typical release flow:
|
|
87
|
+
|
|
88
|
+
1. Update the version in `pyproject.toml`.
|
|
89
|
+
2. Merge to `main`.
|
|
90
|
+
3. Create a tag such as `v0.1.0a1`.
|
|
91
|
+
4. Push the tag.
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
git tag v0.1.0a1
|
|
95
|
+
git push origin v0.1.0a1
|
|
96
|
+
```
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "runlet-harness"
|
|
7
|
+
version = "0.1.0a1"
|
|
8
|
+
description = "Application integration helpers for Runlet."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Runlet Harness contributors" },
|
|
14
|
+
]
|
|
15
|
+
keywords = ["agent", "runtime", "observability", "langfuse", "runlet"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 3 - Alpha",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"runlet>=0.2.0b3",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
langfuse = [
|
|
32
|
+
"langfuse>=4.7,<5",
|
|
33
|
+
]
|
|
34
|
+
dev = [
|
|
35
|
+
"pyright>=1.1.0",
|
|
36
|
+
"ruff>=0.5.0",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.urls]
|
|
40
|
+
Homepage = "https://github.com/DMIAOCHEN/runlet-harness"
|
|
41
|
+
Repository = "https://github.com/DMIAOCHEN/runlet-harness"
|
|
42
|
+
Issues = "https://github.com/DMIAOCHEN/runlet-harness/issues"
|
|
43
|
+
|
|
44
|
+
[tool.hatch.build.targets.wheel]
|
|
45
|
+
packages = ["src/runlet_harness"]
|
|
46
|
+
|
|
47
|
+
[tool.ruff]
|
|
48
|
+
line-length = 100
|
|
49
|
+
target-version = "py310"
|
|
50
|
+
|
|
51
|
+
[tool.ruff.lint]
|
|
52
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
53
|
+
|
|
54
|
+
[tool.pyright]
|
|
55
|
+
pythonVersion = "3.10"
|
|
56
|
+
typeCheckingMode = "strict"
|
|
57
|
+
include = ["src", "tests"]
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
|
|
6
|
+
TRUE_VALUES = {"1", "true", "yes", "y", "on"}
|
|
7
|
+
FALSE_VALUES = {"0", "false", "no", "n", "off"}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def env_str(name: str, *, environ: Mapping[str, str] | None = None) -> str | None:
|
|
11
|
+
source = os.environ if environ is None else environ
|
|
12
|
+
value = source.get(name)
|
|
13
|
+
if value is None:
|
|
14
|
+
return None
|
|
15
|
+
value = value.strip()
|
|
16
|
+
return value or None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def parse_bool(value: str | None, *, name: str) -> bool | None:
|
|
20
|
+
if value is None or value.strip() == "":
|
|
21
|
+
return None
|
|
22
|
+
normalized = value.strip().lower()
|
|
23
|
+
if normalized in TRUE_VALUES:
|
|
24
|
+
return True
|
|
25
|
+
if normalized in FALSE_VALUES:
|
|
26
|
+
return False
|
|
27
|
+
raise ValueError(f"{name} must be a boolean value.")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def env_bool(name: str, *, environ: Mapping[str, str] | None = None) -> bool | None:
|
|
31
|
+
return parse_bool(env_str(name, environ=environ), name=name)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def parse_int(value: str | None, *, name: str) -> int | None:
|
|
35
|
+
if value is None or value.strip() == "":
|
|
36
|
+
return None
|
|
37
|
+
try:
|
|
38
|
+
return int(value)
|
|
39
|
+
except ValueError as error:
|
|
40
|
+
raise ValueError(f"{name} must be an integer.") from error
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def env_int(name: str, *, environ: Mapping[str, str] | None = None) -> int | None:
|
|
44
|
+
return parse_int(env_str(name, environ=environ), name=name)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def parse_float(value: str | None, *, name: str) -> float | None:
|
|
48
|
+
if value is None or value.strip() == "":
|
|
49
|
+
return None
|
|
50
|
+
try:
|
|
51
|
+
return float(value)
|
|
52
|
+
except ValueError as error:
|
|
53
|
+
raise ValueError(f"{name} must be a float.") from error
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def env_float(name: str, *, environ: Mapping[str, str] | None = None) -> float | None:
|
|
57
|
+
return parse_float(env_str(name, environ=environ), name=name)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
from contextlib import suppress
|
|
5
|
+
|
|
6
|
+
from runlet.core import RuntimeEvent
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class BufferedEventSink:
|
|
10
|
+
def __init__(
|
|
11
|
+
self,
|
|
12
|
+
*,
|
|
13
|
+
max_queue_size: int = 1000,
|
|
14
|
+
drop_on_full: bool = True,
|
|
15
|
+
) -> None:
|
|
16
|
+
if max_queue_size < 1:
|
|
17
|
+
raise ValueError("max_queue_size must be at least 1.")
|
|
18
|
+
self.max_queue_size = max_queue_size
|
|
19
|
+
self.drop_on_full = drop_on_full
|
|
20
|
+
self.dropped_events = 0
|
|
21
|
+
self.processed_events = 0
|
|
22
|
+
self.failed_events = 0
|
|
23
|
+
self._queue: asyncio.Queue[RuntimeEvent] | None = None
|
|
24
|
+
self._worker_task: asyncio.Task[None] | None = None
|
|
25
|
+
self._closed = False
|
|
26
|
+
|
|
27
|
+
async def emit(self, event: RuntimeEvent) -> None:
|
|
28
|
+
if self._closed:
|
|
29
|
+
return
|
|
30
|
+
queue = self._ensure_queue()
|
|
31
|
+
if self.drop_on_full:
|
|
32
|
+
try:
|
|
33
|
+
queue.put_nowait(event)
|
|
34
|
+
except asyncio.QueueFull:
|
|
35
|
+
self.dropped_events += 1
|
|
36
|
+
return
|
|
37
|
+
await queue.put(event)
|
|
38
|
+
|
|
39
|
+
async def flush(self) -> None:
|
|
40
|
+
queue = self._queue
|
|
41
|
+
if queue is not None:
|
|
42
|
+
await queue.join()
|
|
43
|
+
await self._flush_backend()
|
|
44
|
+
|
|
45
|
+
async def shutdown(self) -> None:
|
|
46
|
+
if self._closed:
|
|
47
|
+
return
|
|
48
|
+
self._closed = True
|
|
49
|
+
await self.flush()
|
|
50
|
+
task = self._worker_task
|
|
51
|
+
if task is not None:
|
|
52
|
+
task.cancel()
|
|
53
|
+
with suppress(asyncio.CancelledError):
|
|
54
|
+
await task
|
|
55
|
+
self._worker_task = None
|
|
56
|
+
|
|
57
|
+
def _ensure_queue(self) -> asyncio.Queue[RuntimeEvent]:
|
|
58
|
+
if self._queue is None:
|
|
59
|
+
self._queue = asyncio.Queue(maxsize=self.max_queue_size)
|
|
60
|
+
if self._worker_task is None or self._worker_task.done():
|
|
61
|
+
self._worker_task = asyncio.create_task(self._worker())
|
|
62
|
+
return self._queue
|
|
63
|
+
|
|
64
|
+
async def _worker(self) -> None:
|
|
65
|
+
assert self._queue is not None
|
|
66
|
+
while True:
|
|
67
|
+
event = await self._queue.get()
|
|
68
|
+
try:
|
|
69
|
+
await self._process_event(event)
|
|
70
|
+
self.processed_events += 1
|
|
71
|
+
except Exception:
|
|
72
|
+
self.failed_events += 1
|
|
73
|
+
finally:
|
|
74
|
+
self._queue.task_done()
|
|
75
|
+
|
|
76
|
+
async def _process_event(self, event: RuntimeEvent) -> None:
|
|
77
|
+
raise NotImplementedError
|
|
78
|
+
|
|
79
|
+
async def _flush_backend(self) -> None:
|
|
80
|
+
return None
|
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
import random
|
|
6
|
+
from typing import Any, Protocol
|
|
7
|
+
|
|
8
|
+
from runlet.core import RuntimeEvent
|
|
9
|
+
|
|
10
|
+
from runlet_harness.config import env_bool, env_float, env_int, env_str
|
|
11
|
+
from runlet_harness.observability.buffering import BufferedEventSink
|
|
12
|
+
|
|
13
|
+
TERMINAL_EVENTS = {"run.completed", "run.interrupted", "policy.stopped"}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _metadata_map() -> dict[str, Any]:
|
|
17
|
+
return {}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _bool_env_or(name: str, default: bool, environ: Mapping[str, str] | None) -> bool:
|
|
21
|
+
value = env_bool(name, environ=environ)
|
|
22
|
+
return default if value is None else value
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _float_env_or(name: str, default: float, environ: Mapping[str, str] | None) -> float:
|
|
26
|
+
value = env_float(name, environ=environ)
|
|
27
|
+
return default if value is None else value
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _int_env_or(name: str, default: int, environ: Mapping[str, str] | None) -> int:
|
|
31
|
+
value = env_int(name, environ=environ)
|
|
32
|
+
return default if value is None else value
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class LangfuseConfig:
|
|
37
|
+
public_key: str | None = None
|
|
38
|
+
secret_key: str | None = None
|
|
39
|
+
base_url: str | None = None
|
|
40
|
+
enabled: bool | None = None
|
|
41
|
+
sample_rate: float = 1.0
|
|
42
|
+
max_queue_size: int = 1000
|
|
43
|
+
drop_on_full: bool = True
|
|
44
|
+
capture_input: bool = False
|
|
45
|
+
capture_output: bool = True
|
|
46
|
+
capture_reasoning: bool = False
|
|
47
|
+
capture_stream_deltas: bool = False
|
|
48
|
+
capture_tool_arguments: bool = False
|
|
49
|
+
capture_tool_results: bool = False
|
|
50
|
+
max_payload_chars: int = 4000
|
|
51
|
+
flush_on_terminal_event: bool = False
|
|
52
|
+
debug: bool = False
|
|
53
|
+
metadata: dict[str, Any] = field(default_factory=_metadata_map)
|
|
54
|
+
|
|
55
|
+
@classmethod
|
|
56
|
+
def from_env(cls, environ: Mapping[str, str] | None = None) -> "LangfuseConfig | None":
|
|
57
|
+
enabled = env_bool("RUNLET_LANGFUSE_ENABLED", environ=environ)
|
|
58
|
+
public_key = env_str("LANGFUSE_PUBLIC_KEY", environ=environ)
|
|
59
|
+
secret_key = env_str("LANGFUSE_SECRET_KEY", environ=environ)
|
|
60
|
+
base_url = env_str("LANGFUSE_BASE_URL", environ=environ)
|
|
61
|
+
if enabled is False:
|
|
62
|
+
return None
|
|
63
|
+
if enabled is True and (public_key is None or secret_key is None):
|
|
64
|
+
raise ValueError(
|
|
65
|
+
"RUNLET_LANGFUSE_ENABLED=true requires LANGFUSE_PUBLIC_KEY and "
|
|
66
|
+
"LANGFUSE_SECRET_KEY."
|
|
67
|
+
)
|
|
68
|
+
if enabled is None and (public_key is None or secret_key is None):
|
|
69
|
+
return None
|
|
70
|
+
return cls(
|
|
71
|
+
public_key=public_key,
|
|
72
|
+
secret_key=secret_key,
|
|
73
|
+
base_url=base_url,
|
|
74
|
+
enabled=enabled,
|
|
75
|
+
sample_rate=_float_env_or("RUNLET_LANGFUSE_SAMPLE_RATE", 1.0, environ),
|
|
76
|
+
max_queue_size=_int_env_or("RUNLET_LANGFUSE_MAX_QUEUE_SIZE", 1000, environ),
|
|
77
|
+
drop_on_full=_bool_env_or("RUNLET_LANGFUSE_DROP_ON_FULL", True, environ),
|
|
78
|
+
capture_input=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_INPUT", False, environ),
|
|
79
|
+
capture_output=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_OUTPUT", True, environ),
|
|
80
|
+
capture_reasoning=_bool_env_or("RUNLET_LANGFUSE_CAPTURE_REASONING", False, environ),
|
|
81
|
+
capture_stream_deltas=_bool_env_or(
|
|
82
|
+
"RUNLET_LANGFUSE_CAPTURE_STREAM_DELTAS", False, environ
|
|
83
|
+
),
|
|
84
|
+
capture_tool_arguments=_bool_env_or(
|
|
85
|
+
"RUNLET_LANGFUSE_CAPTURE_TOOL_ARGUMENTS", False, environ
|
|
86
|
+
),
|
|
87
|
+
capture_tool_results=_bool_env_or(
|
|
88
|
+
"RUNLET_LANGFUSE_CAPTURE_TOOL_RESULTS", False, environ
|
|
89
|
+
),
|
|
90
|
+
max_payload_chars=_int_env_or("RUNLET_LANGFUSE_MAX_PAYLOAD_CHARS", 4000, environ),
|
|
91
|
+
flush_on_terminal_event=_bool_env_or(
|
|
92
|
+
"RUNLET_LANGFUSE_FLUSH_ON_TERMINAL_EVENT", False, environ
|
|
93
|
+
),
|
|
94
|
+
debug=_bool_env_or("RUNLET_LANGFUSE_DEBUG", False, environ),
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class LangfuseClientProtocol(Protocol):
|
|
99
|
+
def start_observation(self, **kwargs: Any) -> Any:
|
|
100
|
+
...
|
|
101
|
+
|
|
102
|
+
def flush(self) -> Any:
|
|
103
|
+
...
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass
|
|
107
|
+
class _RunState:
|
|
108
|
+
root: Any = None
|
|
109
|
+
model: Any = None
|
|
110
|
+
sampled: bool = True
|
|
111
|
+
tool_stack: list[Any] = field(default_factory=list)
|
|
112
|
+
stream_output: list[str] = field(default_factory=list)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
class LangfuseEventSink(BufferedEventSink):
|
|
116
|
+
def __init__(self, config: LangfuseConfig, *, client: Any | None = None) -> None:
|
|
117
|
+
if config.sample_rate < 0 or config.sample_rate > 1:
|
|
118
|
+
raise ValueError("sample_rate must be between 0 and 1.")
|
|
119
|
+
if config.max_payload_chars < 0:
|
|
120
|
+
raise ValueError("max_payload_chars must not be negative.")
|
|
121
|
+
self.config = config
|
|
122
|
+
self.client = client if client is not None else self._create_client(config)
|
|
123
|
+
self._runs: dict[str, _RunState] = {}
|
|
124
|
+
super().__init__(
|
|
125
|
+
max_queue_size=config.max_queue_size,
|
|
126
|
+
drop_on_full=config.drop_on_full,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
@classmethod
|
|
130
|
+
def from_env(cls, *, client: Any | None = None) -> "LangfuseEventSink | None":
|
|
131
|
+
config = LangfuseConfig.from_env()
|
|
132
|
+
if config is None:
|
|
133
|
+
return None
|
|
134
|
+
return cls(config, client=client)
|
|
135
|
+
|
|
136
|
+
async def _process_event(self, event: RuntimeEvent) -> None:
|
|
137
|
+
state = self._runs.setdefault(event.run_id, _RunState())
|
|
138
|
+
if event.type == "run.started":
|
|
139
|
+
state.sampled = self.config.sample_rate >= 1 or random.random() < self.config.sample_rate
|
|
140
|
+
if not state.sampled:
|
|
141
|
+
return
|
|
142
|
+
state.root = self._start_observation(
|
|
143
|
+
None,
|
|
144
|
+
name="runlet.run",
|
|
145
|
+
as_type="span",
|
|
146
|
+
input=self._payload_value(event, "input") if self.config.capture_input else None,
|
|
147
|
+
metadata=self._metadata(event),
|
|
148
|
+
)
|
|
149
|
+
return
|
|
150
|
+
if not state.sampled:
|
|
151
|
+
if event.type in TERMINAL_EVENTS:
|
|
152
|
+
self._runs.pop(event.run_id, None)
|
|
153
|
+
return
|
|
154
|
+
if event.type == "model.requested":
|
|
155
|
+
state.model = self._start_observation(
|
|
156
|
+
state.root,
|
|
157
|
+
name="runlet.model",
|
|
158
|
+
as_type="generation",
|
|
159
|
+
metadata=self._metadata(event),
|
|
160
|
+
)
|
|
161
|
+
return
|
|
162
|
+
if event.type == "model.completed":
|
|
163
|
+
usage = event.payload.get("usage")
|
|
164
|
+
update: dict[str, Any] = {"metadata": self._metadata(event)}
|
|
165
|
+
if isinstance(usage, int):
|
|
166
|
+
update["usage"] = {"total": usage}
|
|
167
|
+
self._update_observation(state.model, **update)
|
|
168
|
+
self._end_observation(state.model)
|
|
169
|
+
state.model = None
|
|
170
|
+
return
|
|
171
|
+
if event.type == "model.stream.started":
|
|
172
|
+
state.stream_output = []
|
|
173
|
+
state.model = self._start_observation(
|
|
174
|
+
state.root,
|
|
175
|
+
name="runlet.model.stream",
|
|
176
|
+
as_type="generation",
|
|
177
|
+
metadata=self._metadata(event),
|
|
178
|
+
)
|
|
179
|
+
return
|
|
180
|
+
if event.type == "model.stream.delta":
|
|
181
|
+
if self.config.capture_stream_deltas:
|
|
182
|
+
delta = self._payload_value(event, "delta")
|
|
183
|
+
if isinstance(delta, str):
|
|
184
|
+
state.stream_output.append(delta)
|
|
185
|
+
return
|
|
186
|
+
if event.type == "model.stream.completed":
|
|
187
|
+
output = None
|
|
188
|
+
if self.config.capture_stream_deltas:
|
|
189
|
+
output = self._truncate("".join(state.stream_output))
|
|
190
|
+
self._update_observation(state.model, output=output, metadata=self._metadata(event))
|
|
191
|
+
self._end_observation(state.model)
|
|
192
|
+
state.model = None
|
|
193
|
+
state.stream_output = []
|
|
194
|
+
return
|
|
195
|
+
if event.type == "tool.started":
|
|
196
|
+
tool_name = str(event.payload.get("name", "unknown"))
|
|
197
|
+
observation = self._start_observation(
|
|
198
|
+
state.root,
|
|
199
|
+
name=f"runlet.tool.{tool_name}",
|
|
200
|
+
as_type="span",
|
|
201
|
+
metadata=self._metadata(event),
|
|
202
|
+
)
|
|
203
|
+
state.tool_stack.append(observation)
|
|
204
|
+
return
|
|
205
|
+
if event.type == "tool.completed":
|
|
206
|
+
observation = state.tool_stack.pop() if state.tool_stack else None
|
|
207
|
+
self._update_observation(observation, metadata=self._metadata(event))
|
|
208
|
+
self._end_observation(observation)
|
|
209
|
+
return
|
|
210
|
+
if event.type in {"human.requested", "human.responded", "human.response_rejected", "run.resumed"}:
|
|
211
|
+
observation = self._start_observation(
|
|
212
|
+
state.root,
|
|
213
|
+
name=f"runlet.{event.type}",
|
|
214
|
+
as_type="event",
|
|
215
|
+
metadata=self._metadata(event),
|
|
216
|
+
)
|
|
217
|
+
self._end_observation(observation)
|
|
218
|
+
return
|
|
219
|
+
if event.type in TERMINAL_EVENTS:
|
|
220
|
+
update = {"metadata": self._metadata(event)}
|
|
221
|
+
output = self._terminal_output(event)
|
|
222
|
+
if output is not None:
|
|
223
|
+
update["output"] = output
|
|
224
|
+
self._update_observation(state.root, **update)
|
|
225
|
+
self._end_observation(state.root)
|
|
226
|
+
self._runs.pop(event.run_id, None)
|
|
227
|
+
if self.config.flush_on_terminal_event:
|
|
228
|
+
await self._flush_backend()
|
|
229
|
+
|
|
230
|
+
async def _flush_backend(self) -> None:
|
|
231
|
+
result = self.client.flush()
|
|
232
|
+
if hasattr(result, "__await__"):
|
|
233
|
+
await result
|
|
234
|
+
|
|
235
|
+
@staticmethod
|
|
236
|
+
def _create_client(config: LangfuseConfig) -> Any:
|
|
237
|
+
if config.public_key is None or config.secret_key is None:
|
|
238
|
+
raise ValueError("Langfuse public_key and secret_key are required when creating a client.")
|
|
239
|
+
try:
|
|
240
|
+
from langfuse import Langfuse # type: ignore[import-not-found]
|
|
241
|
+
except ImportError as error:
|
|
242
|
+
raise RuntimeError(
|
|
243
|
+
"Langfuse support requires the optional dependency. Install it with "
|
|
244
|
+
"pip install \"runlet-harness[langfuse]\"."
|
|
245
|
+
) from error
|
|
246
|
+
kwargs: dict[str, Any] = {
|
|
247
|
+
"public_key": config.public_key,
|
|
248
|
+
"secret_key": config.secret_key,
|
|
249
|
+
}
|
|
250
|
+
if config.base_url is not None:
|
|
251
|
+
kwargs["base_url"] = config.base_url
|
|
252
|
+
try:
|
|
253
|
+
return Langfuse(**kwargs)
|
|
254
|
+
except TypeError:
|
|
255
|
+
if "base_url" in kwargs:
|
|
256
|
+
kwargs["host"] = kwargs.pop("base_url")
|
|
257
|
+
return Langfuse(**kwargs)
|
|
258
|
+
|
|
259
|
+
def _metadata(self, event: RuntimeEvent) -> dict[str, Any]:
|
|
260
|
+
metadata: dict[str, Any] = {
|
|
261
|
+
"runlet.event_id": event.id,
|
|
262
|
+
"runlet.event_type": event.type,
|
|
263
|
+
"runlet.run_id": event.run_id,
|
|
264
|
+
"runlet.timestamp": event.timestamp.isoformat(),
|
|
265
|
+
"runlet.severity": event.severity,
|
|
266
|
+
**self.config.metadata,
|
|
267
|
+
}
|
|
268
|
+
if event.agent_name is not None:
|
|
269
|
+
metadata["runlet.agent_name"] = event.agent_name
|
|
270
|
+
if event.step_id is not None:
|
|
271
|
+
metadata["runlet.step_id"] = event.step_id
|
|
272
|
+
if event.span_id is not None:
|
|
273
|
+
metadata["runlet.span_id"] = event.span_id
|
|
274
|
+
if event.parent_span_id is not None:
|
|
275
|
+
metadata["runlet.parent_span_id"] = event.parent_span_id
|
|
276
|
+
for key, value in event.attributes.items():
|
|
277
|
+
metadata[f"runlet.attribute.{key}"] = self._sanitize_value(value)
|
|
278
|
+
for key, value in self._safe_payload(event).items():
|
|
279
|
+
metadata[f"runlet.payload.{key}"] = value
|
|
280
|
+
return metadata
|
|
281
|
+
|
|
282
|
+
def _safe_payload(self, event: RuntimeEvent) -> dict[str, Any]:
|
|
283
|
+
payload: dict[str, Any] = {}
|
|
284
|
+
for key, value in event.payload.items():
|
|
285
|
+
if key == "input" and not self.config.capture_input:
|
|
286
|
+
continue
|
|
287
|
+
if key == "output" and not self.config.capture_output:
|
|
288
|
+
continue
|
|
289
|
+
if key == "reasoning" and not self.config.capture_reasoning:
|
|
290
|
+
continue
|
|
291
|
+
if key == "delta" and not self.config.capture_stream_deltas:
|
|
292
|
+
continue
|
|
293
|
+
if key in {"arguments", "tool_arguments"} and not self.config.capture_tool_arguments:
|
|
294
|
+
continue
|
|
295
|
+
if key in {"result", "tool_result"} and not self.config.capture_tool_results:
|
|
296
|
+
continue
|
|
297
|
+
payload[key] = self._sanitize_value(value)
|
|
298
|
+
return payload
|
|
299
|
+
|
|
300
|
+
def _terminal_output(self, event: RuntimeEvent) -> Any | None:
|
|
301
|
+
if not self.config.capture_output:
|
|
302
|
+
return None
|
|
303
|
+
output = self._payload_value(event, "output")
|
|
304
|
+
if output is None:
|
|
305
|
+
return None
|
|
306
|
+
return self._sanitize_value(output)
|
|
307
|
+
|
|
308
|
+
def _payload_value(self, event: RuntimeEvent, key: str) -> Any | None:
|
|
309
|
+
if key not in event.payload:
|
|
310
|
+
return None
|
|
311
|
+
return self._sanitize_value(event.payload[key])
|
|
312
|
+
|
|
313
|
+
def _sanitize_value(self, value: Any) -> Any:
|
|
314
|
+
if isinstance(value, str):
|
|
315
|
+
return self._truncate(value)
|
|
316
|
+
if isinstance(value, (int, float, bool)) or value is None:
|
|
317
|
+
return value
|
|
318
|
+
if isinstance(value, Mapping):
|
|
319
|
+
return {str(key): self._sanitize_value(item) for key, item in value.items()}
|
|
320
|
+
if isinstance(value, (list, tuple)):
|
|
321
|
+
return [self._sanitize_value(item) for item in value]
|
|
322
|
+
return self._truncate(repr(value))
|
|
323
|
+
|
|
324
|
+
def _truncate(self, value: str) -> str:
|
|
325
|
+
limit = self.config.max_payload_chars
|
|
326
|
+
if limit == 0:
|
|
327
|
+
return ""
|
|
328
|
+
if len(value) <= limit:
|
|
329
|
+
return value
|
|
330
|
+
return value[:limit]
|
|
331
|
+
|
|
332
|
+
def _start_observation(self, parent: Any, **kwargs: Any) -> Any:
|
|
333
|
+
target = parent if parent is not None and hasattr(parent, "start_observation") else self.client
|
|
334
|
+
starter = getattr(target, "start_observation")
|
|
335
|
+
kwargs = {key: value for key, value in kwargs.items() if value is not None}
|
|
336
|
+
try:
|
|
337
|
+
return starter(**kwargs)
|
|
338
|
+
except TypeError:
|
|
339
|
+
fallback = dict(kwargs)
|
|
340
|
+
if "as_type" in fallback:
|
|
341
|
+
fallback["type"] = fallback.pop("as_type")
|
|
342
|
+
return starter(**fallback)
|
|
343
|
+
|
|
344
|
+
@staticmethod
|
|
345
|
+
def _update_observation(observation: Any, **kwargs: Any) -> None:
|
|
346
|
+
if observation is None or not hasattr(observation, "update"):
|
|
347
|
+
return
|
|
348
|
+
clean_kwargs = {key: value for key, value in kwargs.items() if value is not None}
|
|
349
|
+
if clean_kwargs:
|
|
350
|
+
observation.update(**clean_kwargs)
|
|
351
|
+
|
|
352
|
+
@staticmethod
|
|
353
|
+
def _end_observation(observation: Any) -> None:
|
|
354
|
+
if observation is None or not hasattr(observation, "end"):
|
|
355
|
+
return
|
|
356
|
+
observation.end()
|
|
File without changes
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import unittest
|
|
3
|
+
|
|
4
|
+
from runlet.core import RuntimeEvent
|
|
5
|
+
from runlet_harness.observability.buffering import BufferedEventSink
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class RecordingSink(BufferedEventSink):
|
|
9
|
+
def __init__(self, **kwargs):
|
|
10
|
+
super().__init__(**kwargs)
|
|
11
|
+
self.events = []
|
|
12
|
+
self.flushed = 0
|
|
13
|
+
self.fail = False
|
|
14
|
+
|
|
15
|
+
async def _process_event(self, event: RuntimeEvent) -> None:
|
|
16
|
+
if self.fail:
|
|
17
|
+
raise RuntimeError("boom")
|
|
18
|
+
self.events.append(event)
|
|
19
|
+
|
|
20
|
+
async def _flush_backend(self) -> None:
|
|
21
|
+
self.flushed += 1
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class BufferedEventSinkTests(unittest.IsolatedAsyncioTestCase):
|
|
25
|
+
async def test_emit_processes_events(self) -> None:
|
|
26
|
+
sink = RecordingSink()
|
|
27
|
+
event = RuntimeEvent(type="run.started", run_id="run_1")
|
|
28
|
+
|
|
29
|
+
await sink.emit(event)
|
|
30
|
+
await sink.flush()
|
|
31
|
+
|
|
32
|
+
self.assertEqual(sink.events, [event])
|
|
33
|
+
self.assertEqual(sink.processed_events, 1)
|
|
34
|
+
await sink.shutdown()
|
|
35
|
+
|
|
36
|
+
async def test_queue_full_drops_when_configured(self) -> None:
|
|
37
|
+
sink = RecordingSink(max_queue_size=1, drop_on_full=True)
|
|
38
|
+
await sink.emit(RuntimeEvent(type="one", run_id="run_1"))
|
|
39
|
+
await sink.emit(RuntimeEvent(type="two", run_id="run_1"))
|
|
40
|
+
|
|
41
|
+
self.assertGreaterEqual(sink.dropped_events, 0)
|
|
42
|
+
await sink.shutdown()
|
|
43
|
+
|
|
44
|
+
async def test_worker_errors_do_not_propagate(self) -> None:
|
|
45
|
+
sink = RecordingSink()
|
|
46
|
+
sink.fail = True
|
|
47
|
+
|
|
48
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
49
|
+
await sink.flush()
|
|
50
|
+
|
|
51
|
+
self.assertEqual(sink.failed_events, 1)
|
|
52
|
+
await sink.shutdown()
|
|
53
|
+
|
|
54
|
+
async def test_shutdown_is_idempotent(self) -> None:
|
|
55
|
+
sink = RecordingSink()
|
|
56
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
57
|
+
await sink.shutdown()
|
|
58
|
+
await sink.shutdown()
|
|
59
|
+
self.assertEqual(sink.flushed, 1)
|
|
60
|
+
|
|
61
|
+
async def test_non_dropping_queue_waits_for_space(self) -> None:
|
|
62
|
+
sink = RecordingSink(max_queue_size=1, drop_on_full=False)
|
|
63
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
64
|
+
await asyncio.wait_for(sink.emit(RuntimeEvent(type="run.completed", run_id="run_1")), 1)
|
|
65
|
+
await sink.shutdown()
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import unittest
|
|
2
|
+
|
|
3
|
+
from runlet_harness.config import env_bool, env_float, env_int, env_str, parse_bool
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ConfigTests(unittest.TestCase):
|
|
7
|
+
def test_env_str_treats_blank_as_missing(self) -> None:
|
|
8
|
+
self.assertIsNone(env_str("MISSING", environ={}))
|
|
9
|
+
self.assertIsNone(env_str("EMPTY", environ={"EMPTY": " "}))
|
|
10
|
+
self.assertEqual(env_str("VALUE", environ={"VALUE": " hello "}), "hello")
|
|
11
|
+
|
|
12
|
+
def test_bool_parsing(self) -> None:
|
|
13
|
+
for value in ("true", "1", "yes", "on"):
|
|
14
|
+
self.assertIs(parse_bool(value, name="FLAG"), True)
|
|
15
|
+
for value in ("false", "0", "no", "off"):
|
|
16
|
+
self.assertIs(parse_bool(value, name="FLAG"), False)
|
|
17
|
+
with self.assertRaisesRegex(ValueError, "FLAG"):
|
|
18
|
+
parse_bool("maybe", name="FLAG")
|
|
19
|
+
|
|
20
|
+
def test_number_parsing(self) -> None:
|
|
21
|
+
self.assertEqual(env_int("COUNT", environ={"COUNT": "5"}), 5)
|
|
22
|
+
self.assertEqual(env_float("RATE", environ={"RATE": "0.25"}), 0.25)
|
|
23
|
+
with self.assertRaisesRegex(ValueError, "COUNT"):
|
|
24
|
+
env_int("COUNT", environ={"COUNT": "x"})
|
|
25
|
+
with self.assertRaisesRegex(ValueError, "RATE"):
|
|
26
|
+
env_float("RATE", environ={"RATE": "x"})
|
|
27
|
+
|
|
28
|
+
def test_env_bool(self) -> None:
|
|
29
|
+
self.assertIs(env_bool("FLAG", environ={"FLAG": "yes"}), True)
|
|
30
|
+
self.assertIs(env_bool("FLAG", environ={"FLAG": "no"}), False)
|
|
31
|
+
self.assertIsNone(env_bool("FLAG", environ={}))
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import unittest
|
|
3
|
+
from unittest.mock import patch
|
|
4
|
+
|
|
5
|
+
from runlet import Agent, Message, Runtime
|
|
6
|
+
from runlet.core import RuntimeEvent
|
|
7
|
+
from runlet.core.models import ModelResponse
|
|
8
|
+
from runlet.testing import FakeModelProvider
|
|
9
|
+
from runlet_harness.observability.langfuse import LangfuseConfig, LangfuseEventSink
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class FakeObservation:
|
|
13
|
+
def __init__(self, client, **kwargs):
|
|
14
|
+
self.client = client
|
|
15
|
+
self.kwargs = kwargs
|
|
16
|
+
self.updates = []
|
|
17
|
+
self.ended = False
|
|
18
|
+
self.children = []
|
|
19
|
+
|
|
20
|
+
def start_observation(self, **kwargs):
|
|
21
|
+
child = FakeObservation(self.client, **kwargs)
|
|
22
|
+
self.children.append(child)
|
|
23
|
+
self.client.observations.append(child)
|
|
24
|
+
return child
|
|
25
|
+
|
|
26
|
+
def update(self, **kwargs):
|
|
27
|
+
self.updates.append(kwargs)
|
|
28
|
+
|
|
29
|
+
def end(self):
|
|
30
|
+
self.ended = True
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class FakeClient:
|
|
34
|
+
def __init__(self):
|
|
35
|
+
self.observations = []
|
|
36
|
+
self.flushed = 0
|
|
37
|
+
self.fail = False
|
|
38
|
+
|
|
39
|
+
def start_observation(self, **kwargs):
|
|
40
|
+
if self.fail:
|
|
41
|
+
raise RuntimeError("langfuse failed")
|
|
42
|
+
observation = FakeObservation(self, **kwargs)
|
|
43
|
+
self.observations.append(observation)
|
|
44
|
+
return observation
|
|
45
|
+
|
|
46
|
+
def flush(self):
|
|
47
|
+
self.flushed += 1
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class LangfuseConfigTests(unittest.TestCase):
|
|
51
|
+
def test_from_env_returns_none_without_keys(self) -> None:
|
|
52
|
+
with patch.dict(os.environ, {}, clear=True):
|
|
53
|
+
self.assertIsNone(LangfuseEventSink.from_env(client=FakeClient()))
|
|
54
|
+
|
|
55
|
+
def test_from_env_can_be_disabled(self) -> None:
|
|
56
|
+
env = {
|
|
57
|
+
"RUNLET_LANGFUSE_ENABLED": "false",
|
|
58
|
+
"LANGFUSE_PUBLIC_KEY": "pk",
|
|
59
|
+
"LANGFUSE_SECRET_KEY": "sk",
|
|
60
|
+
}
|
|
61
|
+
with patch.dict(os.environ, env, clear=True):
|
|
62
|
+
self.assertIsNone(LangfuseEventSink.from_env(client=FakeClient()))
|
|
63
|
+
|
|
64
|
+
def test_forced_enable_requires_keys(self) -> None:
|
|
65
|
+
with patch.dict(os.environ, {"RUNLET_LANGFUSE_ENABLED": "true"}, clear=True):
|
|
66
|
+
with self.assertRaisesRegex(ValueError, "LANGFUSE_PUBLIC_KEY"):
|
|
67
|
+
LangfuseEventSink.from_env(client=FakeClient())
|
|
68
|
+
|
|
69
|
+
def test_env_base_url_is_loaded(self) -> None:
|
|
70
|
+
env = {
|
|
71
|
+
"LANGFUSE_PUBLIC_KEY": "pk",
|
|
72
|
+
"LANGFUSE_SECRET_KEY": "sk",
|
|
73
|
+
"LANGFUSE_BASE_URL": "https://langfuse.example.com",
|
|
74
|
+
}
|
|
75
|
+
with patch.dict(os.environ, env, clear=True):
|
|
76
|
+
config = LangfuseConfig.from_env()
|
|
77
|
+
assert config is not None
|
|
78
|
+
self.assertEqual(config.base_url, "https://langfuse.example.com")
|
|
79
|
+
|
|
80
|
+
def test_env_explicit_zero_values_are_preserved(self) -> None:
|
|
81
|
+
env = {
|
|
82
|
+
"LANGFUSE_PUBLIC_KEY": "pk",
|
|
83
|
+
"LANGFUSE_SECRET_KEY": "sk",
|
|
84
|
+
"RUNLET_LANGFUSE_SAMPLE_RATE": "0",
|
|
85
|
+
"RUNLET_LANGFUSE_MAX_PAYLOAD_CHARS": "0",
|
|
86
|
+
}
|
|
87
|
+
with patch.dict(os.environ, env, clear=True):
|
|
88
|
+
config = LangfuseConfig.from_env()
|
|
89
|
+
assert config is not None
|
|
90
|
+
self.assertEqual(config.sample_rate, 0)
|
|
91
|
+
self.assertEqual(config.max_payload_chars, 0)
|
|
92
|
+
|
|
93
|
+
def test_missing_sdk_error_is_clear(self) -> None:
|
|
94
|
+
with patch("runlet_harness.observability.langfuse.LangfuseEventSink._create_client") as create:
|
|
95
|
+
create.side_effect = RuntimeError('Install it with pip install "runlet-harness[langfuse]".')
|
|
96
|
+
with self.assertRaisesRegex(RuntimeError, "runlet-harness\\[langfuse\\]"):
|
|
97
|
+
LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class LangfuseEventSinkTests(unittest.IsolatedAsyncioTestCase):
|
|
101
|
+
async def test_maps_run_model_tool_and_terminal_events(self) -> None:
|
|
102
|
+
client = FakeClient()
|
|
103
|
+
sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
|
|
104
|
+
|
|
105
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1", agent_name="agent", payload={"input": "secret"}))
|
|
106
|
+
await sink.emit(RuntimeEvent(type="model.requested", run_id="run_1", agent_name="agent"))
|
|
107
|
+
await sink.emit(RuntimeEvent(type="model.completed", run_id="run_1", agent_name="agent", payload={"usage": 7}))
|
|
108
|
+
await sink.emit(RuntimeEvent(type="tool.started", run_id="run_1", agent_name="agent", payload={"name": "lookup"}))
|
|
109
|
+
await sink.emit(RuntimeEvent(type="tool.completed", run_id="run_1", agent_name="agent", payload={"name": "lookup"}))
|
|
110
|
+
await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1", agent_name="agent", payload={"output": "done"}))
|
|
111
|
+
await sink.shutdown()
|
|
112
|
+
|
|
113
|
+
names = [observation.kwargs["name"] for observation in client.observations]
|
|
114
|
+
self.assertIn("runlet.run", names)
|
|
115
|
+
self.assertIn("runlet.model", names)
|
|
116
|
+
self.assertIn("runlet.tool.lookup", names)
|
|
117
|
+
root = client.observations[0]
|
|
118
|
+
self.assertNotIn("input", root.kwargs)
|
|
119
|
+
self.assertTrue(root.ended)
|
|
120
|
+
self.assertEqual(root.updates[-1]["output"], "done")
|
|
121
|
+
|
|
122
|
+
async def test_stream_delta_is_ignored_by_default(self) -> None:
|
|
123
|
+
client = FakeClient()
|
|
124
|
+
sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
|
|
125
|
+
|
|
126
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
127
|
+
await sink.emit(RuntimeEvent(type="model.stream.started", run_id="run_1"))
|
|
128
|
+
await sink.emit(RuntimeEvent(type="model.stream.delta", run_id="run_1", payload={"delta": "private"}))
|
|
129
|
+
await sink.emit(RuntimeEvent(type="model.stream.completed", run_id="run_1"))
|
|
130
|
+
await sink.shutdown()
|
|
131
|
+
|
|
132
|
+
stream = [observation for observation in client.observations if observation.kwargs["name"] == "runlet.model.stream"][0]
|
|
133
|
+
self.assertNotIn("output", stream.updates[-1])
|
|
134
|
+
|
|
135
|
+
async def test_stream_delta_can_be_captured_and_truncated(self) -> None:
|
|
136
|
+
client = FakeClient()
|
|
137
|
+
config = LangfuseConfig(
|
|
138
|
+
public_key="pk",
|
|
139
|
+
secret_key="sk",
|
|
140
|
+
capture_stream_deltas=True,
|
|
141
|
+
max_payload_chars=4,
|
|
142
|
+
)
|
|
143
|
+
sink = LangfuseEventSink(config, client=client)
|
|
144
|
+
|
|
145
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
146
|
+
await sink.emit(RuntimeEvent(type="model.stream.started", run_id="run_1"))
|
|
147
|
+
await sink.emit(RuntimeEvent(type="model.stream.delta", run_id="run_1", payload={"delta": "private"}))
|
|
148
|
+
await sink.emit(RuntimeEvent(type="model.stream.completed", run_id="run_1"))
|
|
149
|
+
await sink.shutdown()
|
|
150
|
+
|
|
151
|
+
stream = [observation for observation in client.observations if observation.kwargs["name"] == "runlet.model.stream"][0]
|
|
152
|
+
self.assertEqual(stream.updates[-1]["output"], "priv")
|
|
153
|
+
|
|
154
|
+
async def test_client_errors_do_not_fail_runtime(self) -> None:
|
|
155
|
+
client = FakeClient()
|
|
156
|
+
client.fail = True
|
|
157
|
+
sink = LangfuseEventSink(LangfuseConfig(public_key="pk", secret_key="sk"), client=client)
|
|
158
|
+
model = FakeModelProvider([ModelResponse(message=Message.assistant("ok"))])
|
|
159
|
+
agent = Agent(name="assistant", instructions="Help.", model=model)
|
|
160
|
+
|
|
161
|
+
result = await Runtime(event_sink=sink).run(agent, "hi")
|
|
162
|
+
await sink.shutdown()
|
|
163
|
+
|
|
164
|
+
self.assertEqual(result.output, "ok")
|
|
165
|
+
self.assertGreater(sink.failed_events, 0)
|
|
166
|
+
|
|
167
|
+
async def test_sample_rate_zero_skips_observations(self) -> None:
|
|
168
|
+
client = FakeClient()
|
|
169
|
+
sink = LangfuseEventSink(
|
|
170
|
+
LangfuseConfig(public_key="pk", secret_key="sk", sample_rate=0),
|
|
171
|
+
client=client,
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
175
|
+
await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1"))
|
|
176
|
+
await sink.shutdown()
|
|
177
|
+
|
|
178
|
+
self.assertEqual(client.observations, [])
|
|
179
|
+
|
|
180
|
+
async def test_flush_on_terminal_event(self) -> None:
|
|
181
|
+
client = FakeClient()
|
|
182
|
+
config = LangfuseConfig(public_key="pk", secret_key="sk", flush_on_terminal_event=True)
|
|
183
|
+
sink = LangfuseEventSink(config, client=client)
|
|
184
|
+
|
|
185
|
+
await sink.emit(RuntimeEvent(type="run.started", run_id="run_1"))
|
|
186
|
+
await sink.emit(RuntimeEvent(type="run.completed", run_id="run_1"))
|
|
187
|
+
await sink.shutdown()
|
|
188
|
+
|
|
189
|
+
self.assertGreaterEqual(client.flushed, 1)
|