story-test 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- story_test-1.2.0/LICENSE +21 -0
- story_test-1.2.0/PKG-INFO +176 -0
- story_test-1.2.0/README.md +156 -0
- story_test-1.2.0/pyproject.toml +41 -0
- story_test-1.2.0/schemas/v1/tests.schema.json +33 -0
- story_test-1.2.0/setup.cfg +4 -0
- story_test-1.2.0/src/story_test/__init__.py +3 -0
- story_test-1.2.0/src/story_test/cli.py +186 -0
- story_test-1.2.0/src/story_test.egg-info/PKG-INFO +176 -0
- story_test-1.2.0/src/story_test.egg-info/SOURCES.txt +13 -0
- story_test-1.2.0/src/story_test.egg-info/dependency_links.txt +1 -0
- story_test-1.2.0/src/story_test.egg-info/entry_points.txt +2 -0
- story_test-1.2.0/src/story_test.egg-info/requires.txt +3 -0
- story_test-1.2.0/src/story_test.egg-info/top_level.txt +1 -0
- story_test-1.2.0/test/test_story_test.py +64 -0
story_test-1.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 legomb
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: story-test
|
|
3
|
+
Version: 1.2.0
|
|
4
|
+
Summary: Evaluate story assertions with a local Ollama model.
|
|
5
|
+
Author: legomb
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/legomb/story-test
|
|
8
|
+
Project-URL: Issues, https://github.com/legomb/story-test/issues
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Requires-Dist: jsonschema>=4.0
|
|
17
|
+
Requires-Dist: ollama>=0.6
|
|
18
|
+
Requires-Dist: PyYAML>=6.0
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# Story test
|
|
22
|
+
|
|
23
|
+
[](https://github.com/legomb/story-test/actions/workflows/build.yml)
|
|
24
|
+
|
|
25
|
+
CLI tool that runs tests against a story.
|
|
26
|
+
|
|
27
|
+
Lets you define a list of checks you expect from your story (e.g. "The hero wins in the end"), and determines whether each one passes or fails.
|
|
28
|
+
|
|
29
|
+
It does not replace a human editor, but it's a great tool to aid in the editing phase, for both writers and editors:
|
|
30
|
+
|
|
31
|
+
- Make sure your story's main points are addressed while editing your story.
|
|
32
|
+
- Build and grow a repository with standard tests you want to run on manuscripts, and make specific tests for specific genres, etc.
|
|
33
|
+
|
|
34
|
+
Uses a local [Ollama](https://ollama.com/) model.
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
Create a test file containing assertions about a story:
|
|
39
|
+
|
|
40
|
+
```yaml
|
|
41
|
+
tests:
|
|
42
|
+
- name: author
|
|
43
|
+
assertion: The story was written by Edgar Allan Poe.
|
|
44
|
+
- name: ending
|
|
45
|
+
assertion: The narrator confesses at the end of the story.
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Run the tests against one or more Markdown files:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
story-test story.tests.yml story.md
|
|
52
|
+
story-test story.tests.yml chapter-1.md chapter-2.md
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Failed assertions are reported without failing the process by default. For CI, use strict mode:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
story-test story.tests.yml story.md --fail-on-test-failure
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## 🚀 Getting Started
|
|
62
|
+
|
|
63
|
+
This repo uses **direnv**, **Devbox**, **Taskfile**, and **pre-commit** for a reproducible dev environment and automatic schema/YAML validation.
|
|
64
|
+
|
|
65
|
+
### Setup
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
# Automatically enter devbox via direnv (if available)
|
|
69
|
+
direnv allow
|
|
70
|
+
|
|
71
|
+
# Enter dev environment
|
|
72
|
+
devbox shell
|
|
73
|
+
|
|
74
|
+
# Install pre-commit hooks
|
|
75
|
+
task pre-commit:install
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Tasks
|
|
79
|
+
|
|
80
|
+
Run `task` to see a list of available tasks.
|
|
81
|
+
|
|
82
|
+
Install the development dependencies with:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
task environment:dev:install
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
For a local user installation, run the bootstrap script from this repository:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
sh install.sh
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
This creates an isolated Python environment, installs the `story-test` command,
|
|
95
|
+
and pulls the default Ollama model. The installed command can then be used from
|
|
96
|
+
any directory:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
story-test path/to/story.tests.yml path/to/story.md
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Set `STORY_TEST_MODEL` before running the installer to use another model:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
STORY_TEST_MODEL=qwen3:30b-a3b sh install.sh
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Install Ollama separately, then download the default model through Task:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
task environment:ollama:install
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
To use a model already installed locally:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
OLLAMA_MODEL=qwen3:30b-a3b task environment:ollama:install
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Run the sample story tests. Failed story assertions are reported but do not
|
|
121
|
+
fail the task by default:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
task test:example-story
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
The local model is `qwen3:8b` by default. Set `STORY_TEST_MODEL` to use another
|
|
128
|
+
model already installed in Ollama.
|
|
129
|
+
|
|
130
|
+
The model can be changed with `STORY_TEST_MODEL`, and the context window can be
|
|
131
|
+
changed with `STORY_TEST_CONTEXT_LENGTH`.
|
|
132
|
+
|
|
133
|
+
To make failed story assertions fail the task, use the strict variant:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
task test:example-story:strict
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Run the complete local validation suite:
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
task test:all
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
This runs schema validation, Python unit tests, and the sample story tests.
|
|
146
|
+
|
|
147
|
+
The GitHub Actions workflow installs Ollama and pulls `qwen3:8b` automatically.
|
|
148
|
+
Qwen open-weight models are Apache 2.0 licensed and Ollama is MIT licensed;
|
|
149
|
+
always review the license for the exact model tag you deploy.
|
|
150
|
+
|
|
151
|
+
Run formatting and pre-commit checks with:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
task format:check
|
|
155
|
+
task pre-commit:run
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Before publishing a release, build and validate both distribution formats:
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
task package:check
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
This creates the wheel and source archive under `dist/` and validates them with
|
|
165
|
+
Twine. Increment the version in `pyproject.toml` before building a new release.
|
|
166
|
+
|
|
167
|
+
## Features
|
|
168
|
+
|
|
169
|
+
- [x] JSON structure validation using `jq`
|
|
170
|
+
- [x] Schema validation using `check-jsonschema` (temporarily disabled)
|
|
171
|
+
- [x] CI/CD integration with GitHub Actions
|
|
172
|
+
- [x] Versioning schemas with directories like `schemas/v1`, `schemas/v2`
|
|
173
|
+
- [x] Documentation with inline schema descriptions
|
|
174
|
+
- [x] Code formatting using `prettier` or `jq`
|
|
175
|
+
- [ ] Documentation with `README` or extended docs folder (pending)
|
|
176
|
+
- [ ] Schema hosting via `$id` URLs or SchemaStore (pending)
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# Story test
|
|
2
|
+
|
|
3
|
+
[](https://github.com/legomb/story-test/actions/workflows/build.yml)
|
|
4
|
+
|
|
5
|
+
CLI tool that runs tests against a story.
|
|
6
|
+
|
|
7
|
+
Lets you define a list of checks you expect from your story (e.g. "The hero wins in the end"), and determines whether each one passes or fails.
|
|
8
|
+
|
|
9
|
+
It does not replace a human editor, but it's a great tool to aid in the editing phase, for both writers and editors:
|
|
10
|
+
|
|
11
|
+
- Make sure your story's main points are addressed while editing your story.
|
|
12
|
+
- Build and grow a repository with standard tests you want to run on manuscripts, and make specific tests for specific genres, etc.
|
|
13
|
+
|
|
14
|
+
Uses a local [Ollama](https://ollama.com/) model.
|
|
15
|
+
|
|
16
|
+
## Usage
|
|
17
|
+
|
|
18
|
+
Create a test file containing assertions about a story:
|
|
19
|
+
|
|
20
|
+
```yaml
|
|
21
|
+
tests:
|
|
22
|
+
- name: author
|
|
23
|
+
assertion: The story was written by Edgar Allan Poe.
|
|
24
|
+
- name: ending
|
|
25
|
+
assertion: The narrator confesses at the end of the story.
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Run the tests against one or more Markdown files:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
story-test story.tests.yml story.md
|
|
32
|
+
story-test story.tests.yml chapter-1.md chapter-2.md
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Failed assertions are reported without failing the process by default. For CI, use strict mode:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
story-test story.tests.yml story.md --fail-on-test-failure
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## 🚀 Getting Started
|
|
42
|
+
|
|
43
|
+
This repo uses **direnv**, **Devbox**, **Taskfile**, and **pre-commit** for a reproducible dev environment and automatic schema/YAML validation.
|
|
44
|
+
|
|
45
|
+
### Setup
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
# Automatically enter devbox via direnv (if available)
|
|
49
|
+
direnv allow
|
|
50
|
+
|
|
51
|
+
# Enter dev environment
|
|
52
|
+
devbox shell
|
|
53
|
+
|
|
54
|
+
# Install pre-commit hooks
|
|
55
|
+
task pre-commit:install
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Tasks
|
|
59
|
+
|
|
60
|
+
Run `task` to see a list of available tasks.
|
|
61
|
+
|
|
62
|
+
Install the development dependencies with:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
task environment:dev:install
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
For a local user installation, run the bootstrap script from this repository:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
sh install.sh
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
This creates an isolated Python environment, installs the `story-test` command,
|
|
75
|
+
and pulls the default Ollama model. The installed command can then be used from
|
|
76
|
+
any directory:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
story-test path/to/story.tests.yml path/to/story.md
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Set `STORY_TEST_MODEL` before running the installer to use another model:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
STORY_TEST_MODEL=qwen3:30b-a3b sh install.sh
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Install Ollama separately, then download the default model through Task:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
task environment:ollama:install
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
To use a model already installed locally:
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
OLLAMA_MODEL=qwen3:30b-a3b task environment:ollama:install
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Run the sample story tests. Failed story assertions are reported but do not
|
|
101
|
+
fail the task by default:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
task test:example-story
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
The local model is `qwen3:8b` by default. Set `STORY_TEST_MODEL` to use another
|
|
108
|
+
model already installed in Ollama.
|
|
109
|
+
|
|
110
|
+
The model can be changed with `STORY_TEST_MODEL`, and the context window can be
|
|
111
|
+
changed with `STORY_TEST_CONTEXT_LENGTH`.
|
|
112
|
+
|
|
113
|
+
To make failed story assertions fail the task, use the strict variant:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
task test:example-story:strict
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Run the complete local validation suite:
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
task test:all
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
This runs schema validation, Python unit tests, and the sample story tests.
|
|
126
|
+
|
|
127
|
+
The GitHub Actions workflow installs Ollama and pulls `qwen3:8b` automatically.
|
|
128
|
+
Qwen open-weight models are Apache 2.0 licensed and Ollama is MIT licensed;
|
|
129
|
+
always review the license for the exact model tag you deploy.
|
|
130
|
+
|
|
131
|
+
Run formatting and pre-commit checks with:
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
task format:check
|
|
135
|
+
task pre-commit:run
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Before publishing a release, build and validate both distribution formats:
|
|
139
|
+
|
|
140
|
+
```bash
|
|
141
|
+
task package:check
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
This creates the wheel and source archive under `dist/` and validates them with
|
|
145
|
+
Twine. Increment the version in `pyproject.toml` before building a new release.
|
|
146
|
+
|
|
147
|
+
## Features
|
|
148
|
+
|
|
149
|
+
- [x] JSON structure validation using `jq`
|
|
150
|
+
- [x] Schema validation using `check-jsonschema` (temporarily disabled)
|
|
151
|
+
- [x] CI/CD integration with GitHub Actions
|
|
152
|
+
- [x] Versioning schemas with directories like `schemas/v1`, `schemas/v2`
|
|
153
|
+
- [x] Documentation with inline schema descriptions
|
|
154
|
+
- [x] Code formatting using `prettier` or `jq`
|
|
155
|
+
- [ ] Documentation with `README` or extended docs folder (pending)
|
|
156
|
+
- [ ] Schema hosting via `$id` URLs or SchemaStore (pending)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0.3"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "story-test"
|
|
7
|
+
version = "1.2.0"
|
|
8
|
+
description = "Evaluate story assertions with a local Ollama model."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
authors = [{name = "legomb"}]
|
|
12
|
+
license = "MIT"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"jsonschema>=4.0",
|
|
22
|
+
"ollama>=0.6",
|
|
23
|
+
"PyYAML>=6.0",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.scripts]
|
|
27
|
+
story-test = "story_test:main"
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/legomb/story-test"
|
|
31
|
+
Issues = "https://github.com/legomb/story-test/issues"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools]
|
|
34
|
+
package-dir = {"" = "src"}
|
|
35
|
+
packages = ["story_test"]
|
|
36
|
+
|
|
37
|
+
[tool.setuptools.data-files]
|
|
38
|
+
"share/story-test/schemas/v1" = ["schemas/v1/tests.schema.json"]
|
|
39
|
+
|
|
40
|
+
[tool.pytest.ini_options]
|
|
41
|
+
pythonpath = ["src"]
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$id": "https://example.com/story-tests.schema.json",
|
|
3
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
4
|
+
"title": "Story tests schema",
|
|
5
|
+
"description": "Schema for a list of assertions about a story.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": ["tests"],
|
|
9
|
+
"properties": {
|
|
10
|
+
"tests": {
|
|
11
|
+
"type": "array",
|
|
12
|
+
"description": "A list of story assertions to evaluate.",
|
|
13
|
+
"minItems": 1,
|
|
14
|
+
"items": {
|
|
15
|
+
"type": "object",
|
|
16
|
+
"additionalProperties": false,
|
|
17
|
+
"required": ["name", "assertion"],
|
|
18
|
+
"properties": {
|
|
19
|
+
"name": {
|
|
20
|
+
"type": "string",
|
|
21
|
+
"description": "A unique name for the test.",
|
|
22
|
+
"minLength": 1
|
|
23
|
+
},
|
|
24
|
+
"assertion": {
|
|
25
|
+
"type": "string",
|
|
26
|
+
"description": "A claim about the story that can be evaluated as passing or failing.",
|
|
27
|
+
"minLength": 1
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Evaluate story assertions with a local Ollama model."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Iterable
|
|
11
|
+
|
|
12
|
+
from ollama import Client, ResponseError
|
|
13
|
+
import yaml
|
|
14
|
+
from jsonschema import validate
|
|
15
|
+
|
|
16
|
+
SOURCE_SCHEMA_PATH = (
|
|
17
|
+
Path(__file__).resolve().parents[2] / "schemas/v1/tests.schema.json"
|
|
18
|
+
)
|
|
19
|
+
INSTALLED_SCHEMA_PATH = (
|
|
20
|
+
Path(sys.prefix) / "share/story-test/schemas/v1/tests.schema.json"
|
|
21
|
+
)
|
|
22
|
+
DEFAULT_MODEL = "qwen3:8b"
|
|
23
|
+
DEFAULT_OLLAMA_HOST = "http://127.0.0.1:11434"
|
|
24
|
+
DEFAULT_CONTEXT_LENGTH = 32768
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class OllamaRunner:
|
|
28
|
+
def __init__(self, model: str, host: str, context_length: int) -> None:
|
|
29
|
+
self.model = model
|
|
30
|
+
self.context_length = context_length
|
|
31
|
+
self.client = Client(host=host)
|
|
32
|
+
|
|
33
|
+
def predict(
|
|
34
|
+
self, state: str, questions: dict[str, dict[str, Any]]
|
|
35
|
+
) -> dict[str, Any]:
|
|
36
|
+
answers = {}
|
|
37
|
+
for question_id in questions:
|
|
38
|
+
try:
|
|
39
|
+
response = self.client.chat(
|
|
40
|
+
model=self.model,
|
|
41
|
+
messages=[
|
|
42
|
+
{
|
|
43
|
+
"role": "system",
|
|
44
|
+
"content": (
|
|
45
|
+
"Evaluate the assertion against the story. "
|
|
46
|
+
"Return true only when the story supports it."
|
|
47
|
+
),
|
|
48
|
+
},
|
|
49
|
+
{"role": "user", "content": state},
|
|
50
|
+
],
|
|
51
|
+
format={
|
|
52
|
+
"type": "object",
|
|
53
|
+
"properties": {"supported": {"type": "boolean"}},
|
|
54
|
+
"required": ["supported"],
|
|
55
|
+
},
|
|
56
|
+
options={"num_ctx": self.context_length, "temperature": 0},
|
|
57
|
+
)
|
|
58
|
+
except ResponseError as error:
|
|
59
|
+
if error.status_code == 404:
|
|
60
|
+
raise RuntimeError(
|
|
61
|
+
f"Ollama model {self.model!r} is not installed. "
|
|
62
|
+
f"Run `OLLAMA_MODEL={self.model} task environment:ollama:install`."
|
|
63
|
+
) from error
|
|
64
|
+
raise
|
|
65
|
+
content = response.message.content
|
|
66
|
+
parsed = json.loads(content)
|
|
67
|
+
answers[question_id] = {"supported": parsed["supported"]}
|
|
68
|
+
return {"answers": answers}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def load_tests(path: Path) -> list[dict[str, str]]:
|
|
72
|
+
with path.open(encoding="utf-8") as file:
|
|
73
|
+
document = yaml.safe_load(file)
|
|
74
|
+
|
|
75
|
+
schema_path = (
|
|
76
|
+
SOURCE_SCHEMA_PATH if SOURCE_SCHEMA_PATH.exists() else INSTALLED_SCHEMA_PATH
|
|
77
|
+
)
|
|
78
|
+
with schema_path.open(encoding="utf-8") as file:
|
|
79
|
+
schema = yaml.safe_load(file)
|
|
80
|
+
validate(document, schema)
|
|
81
|
+
return document["tests"]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def load_story(paths: Iterable[Path]) -> str:
|
|
85
|
+
sections = []
|
|
86
|
+
for path in paths:
|
|
87
|
+
sections.append(path.read_text(encoding="utf-8"))
|
|
88
|
+
return "\n\n".join(sections)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _answer_is_true(result: Any, question_id: str) -> bool:
|
|
92
|
+
answers = result.get("answers") if isinstance(result, dict) else None
|
|
93
|
+
if not isinstance(answers, dict) or question_id not in answers:
|
|
94
|
+
raise ValueError(f"Ollama did not return an answer for {question_id!r}")
|
|
95
|
+
|
|
96
|
+
answer = answers[question_id]
|
|
97
|
+
if isinstance(answer, dict) and "supported" in answer:
|
|
98
|
+
return answer["supported"] is True
|
|
99
|
+
if isinstance(answer, dict) and "choice" in answer:
|
|
100
|
+
return answer["choice"] == "true"
|
|
101
|
+
if isinstance(answer, dict) and "noul" in answer:
|
|
102
|
+
answer = answer["noul"]
|
|
103
|
+
if isinstance(answer, bool):
|
|
104
|
+
return answer
|
|
105
|
+
if isinstance(answer, (int, float)):
|
|
106
|
+
return answer >= 0.5
|
|
107
|
+
raise ValueError(f"Unexpected Ollama answer for {question_id!r}: {answer!r}")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def run_tests(
|
|
111
|
+
tests_path: Path,
|
|
112
|
+
story_paths: Iterable[Path],
|
|
113
|
+
runner: Any | None = None,
|
|
114
|
+
model: str = DEFAULT_MODEL,
|
|
115
|
+
ollama_host: str = DEFAULT_OLLAMA_HOST,
|
|
116
|
+
context_length: int = DEFAULT_CONTEXT_LENGTH,
|
|
117
|
+
) -> list[dict[str, Any]]:
|
|
118
|
+
tests = load_tests(tests_path)
|
|
119
|
+
story = load_story(story_paths)
|
|
120
|
+
runner = runner or OllamaRunner(model, ollama_host, context_length)
|
|
121
|
+
results = []
|
|
122
|
+
for test in tests:
|
|
123
|
+
questions = {
|
|
124
|
+
test["name"]: {
|
|
125
|
+
"type": "boolean",
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
state = f"Assertion: {test['assertion']}\n\nStory:\n{story}"
|
|
129
|
+
result = runner.predict(state, questions)
|
|
130
|
+
results.append(
|
|
131
|
+
{
|
|
132
|
+
"name": test["name"],
|
|
133
|
+
"passed": _answer_is_true(result, test["name"]),
|
|
134
|
+
}
|
|
135
|
+
)
|
|
136
|
+
return results
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def exit_code(results: list[dict[str, Any]], fail_on_test_failure: bool) -> int:
|
|
140
|
+
if fail_on_test_failure and any(not result["passed"] for result in results):
|
|
141
|
+
return 1
|
|
142
|
+
return 0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def main() -> int:
|
|
146
|
+
parser = argparse.ArgumentParser(
|
|
147
|
+
description="Run Ollama assertions against Markdown stories."
|
|
148
|
+
)
|
|
149
|
+
parser.add_argument(
|
|
150
|
+
"tests", type=Path, help="YAML file matching the story tests schema"
|
|
151
|
+
)
|
|
152
|
+
parser.add_argument(
|
|
153
|
+
"stories", type=Path, nargs="+", help="One or more Markdown story files"
|
|
154
|
+
)
|
|
155
|
+
parser.add_argument(
|
|
156
|
+
"--fail-on-test-failure",
|
|
157
|
+
action="store_true",
|
|
158
|
+
help="Exit with status 1 when any story test fails",
|
|
159
|
+
)
|
|
160
|
+
parser.add_argument("--model", default=os.getenv("STORY_TEST_MODEL", DEFAULT_MODEL))
|
|
161
|
+
parser.add_argument(
|
|
162
|
+
"--ollama-host",
|
|
163
|
+
default=os.getenv("OLLAMA_HOST", DEFAULT_OLLAMA_HOST),
|
|
164
|
+
)
|
|
165
|
+
parser.add_argument(
|
|
166
|
+
"--context-length",
|
|
167
|
+
type=int,
|
|
168
|
+
default=int(os.getenv("STORY_TEST_CONTEXT_LENGTH", DEFAULT_CONTEXT_LENGTH)),
|
|
169
|
+
)
|
|
170
|
+
args = parser.parse_args()
|
|
171
|
+
|
|
172
|
+
results = run_tests(
|
|
173
|
+
args.tests,
|
|
174
|
+
args.stories,
|
|
175
|
+
model=args.model,
|
|
176
|
+
ollama_host=args.ollama_host,
|
|
177
|
+
context_length=args.context_length,
|
|
178
|
+
)
|
|
179
|
+
for result in results:
|
|
180
|
+
status = "PASS" if result["passed"] else "FAIL"
|
|
181
|
+
print(f"Test [{result['name']}]: {status}")
|
|
182
|
+
return exit_code(results, args.fail_on_test_failure)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
if __name__ == "__main__":
|
|
186
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: story-test
|
|
3
|
+
Version: 1.2.0
|
|
4
|
+
Summary: Evaluate story assertions with a local Ollama model.
|
|
5
|
+
Author: legomb
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/legomb/story-test
|
|
8
|
+
Project-URL: Issues, https://github.com/legomb/story-test/issues
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Requires-Dist: jsonschema>=4.0
|
|
17
|
+
Requires-Dist: ollama>=0.6
|
|
18
|
+
Requires-Dist: PyYAML>=6.0
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# Story test
|
|
22
|
+
|
|
23
|
+
[](https://github.com/legomb/story-test/actions/workflows/build.yml)
|
|
24
|
+
|
|
25
|
+
CLI tool that runs tests against a story.
|
|
26
|
+
|
|
27
|
+
Lets you define a list of checks you expect from your story (e.g. "The hero wins in the end"), and determines whether each one passes or fails.
|
|
28
|
+
|
|
29
|
+
It does not replace a human editor, but it's a great tool to aid in the editing phase, for both writers and editors:
|
|
30
|
+
|
|
31
|
+
- Make sure your story's main points are addressed while editing your story.
|
|
32
|
+
- Build and grow a repository with standard tests you want to run on manuscripts, and make specific tests for specific genres, etc.
|
|
33
|
+
|
|
34
|
+
Uses a local [Ollama](https://ollama.com/) model.
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
Create a test file containing assertions about a story:
|
|
39
|
+
|
|
40
|
+
```yaml
|
|
41
|
+
tests:
|
|
42
|
+
- name: author
|
|
43
|
+
assertion: The story was written by Edgar Allan Poe.
|
|
44
|
+
- name: ending
|
|
45
|
+
assertion: The narrator confesses at the end of the story.
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Run the tests against one or more Markdown files:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
story-test story.tests.yml story.md
|
|
52
|
+
story-test story.tests.yml chapter-1.md chapter-2.md
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Failed assertions are reported without failing the process by default. For CI, use strict mode:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
story-test story.tests.yml story.md --fail-on-test-failure
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## 🚀 Getting Started
|
|
62
|
+
|
|
63
|
+
This repo uses **direnv**, **Devbox**, **Taskfile**, and **pre-commit** for a reproducible dev environment and automatic schema/YAML validation.
|
|
64
|
+
|
|
65
|
+
### Setup
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
# Automatically enter devbox via direnv (if available)
|
|
69
|
+
direnv allow
|
|
70
|
+
|
|
71
|
+
# Enter dev environment
|
|
72
|
+
devbox shell
|
|
73
|
+
|
|
74
|
+
# Install pre-commit hooks
|
|
75
|
+
task pre-commit:install
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Tasks
|
|
79
|
+
|
|
80
|
+
Run `task` to see a list of available tasks.
|
|
81
|
+
|
|
82
|
+
Install the development dependencies with:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
task environment:dev:install
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
For a local user installation, run the bootstrap script from this repository:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
sh install.sh
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
This creates an isolated Python environment, installs the `story-test` command,
|
|
95
|
+
and pulls the default Ollama model. The installed command can then be used from
|
|
96
|
+
any directory:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
story-test path/to/story.tests.yml path/to/story.md
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Set `STORY_TEST_MODEL` before running the installer to use another model:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
STORY_TEST_MODEL=qwen3:30b-a3b sh install.sh
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Install Ollama separately, then download the default model through Task:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
task environment:ollama:install
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
To use a model already installed locally:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
OLLAMA_MODEL=qwen3:30b-a3b task environment:ollama:install
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Run the sample story tests. Failed story assertions are reported but do not
|
|
121
|
+
fail the task by default:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
task test:example-story
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
The local model is `qwen3:8b` by default. Set `STORY_TEST_MODEL` to use another
|
|
128
|
+
model already installed in Ollama.
|
|
129
|
+
|
|
130
|
+
The model can be changed with `STORY_TEST_MODEL`, and the context window can be
|
|
131
|
+
changed with `STORY_TEST_CONTEXT_LENGTH`.
|
|
132
|
+
|
|
133
|
+
To make failed story assertions fail the task, use the strict variant:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
task test:example-story:strict
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Run the complete local validation suite:
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
task test:all
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
This runs schema validation, Python unit tests, and the sample story tests.
|
|
146
|
+
|
|
147
|
+
The GitHub Actions workflow installs Ollama and pulls `qwen3:8b` automatically.
|
|
148
|
+
Qwen open-weight models are Apache 2.0 licensed and Ollama is MIT licensed;
|
|
149
|
+
always review the license for the exact model tag you deploy.
|
|
150
|
+
|
|
151
|
+
Run formatting and pre-commit checks with:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
task format:check
|
|
155
|
+
task pre-commit:run
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Before publishing a release, build and validate both distribution formats:
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
task package:check
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
This creates the wheel and source archive under `dist/` and validates them with
|
|
165
|
+
Twine. Increment the version in `pyproject.toml` before building a new release.
|
|
166
|
+
|
|
167
|
+
## Features
|
|
168
|
+
|
|
169
|
+
- [x] JSON structure validation using `jq`
|
|
170
|
+
- [x] Schema validation using `check-jsonschema` (temporarily disabled)
|
|
171
|
+
- [x] CI/CD integration with GitHub Actions
|
|
172
|
+
- [x] Versioning schemas with directories like `schemas/v1`, `schemas/v2`
|
|
173
|
+
- [x] Documentation with inline schema descriptions
|
|
174
|
+
- [x] Code formatting using `prettier` or `jq`
|
|
175
|
+
- [ ] Documentation with `README` or extended docs folder (pending)
|
|
176
|
+
- [ ] Schema hosting via `$id` URLs or SchemaStore (pending)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
schemas/v1/tests.schema.json
|
|
5
|
+
src/story_test/__init__.py
|
|
6
|
+
src/story_test/cli.py
|
|
7
|
+
src/story_test.egg-info/PKG-INFO
|
|
8
|
+
src/story_test.egg-info/SOURCES.txt
|
|
9
|
+
src/story_test.egg-info/dependency_links.txt
|
|
10
|
+
src/story_test.egg-info/entry_points.txt
|
|
11
|
+
src/story_test.egg-info/requires.txt
|
|
12
|
+
src/story_test.egg-info/top_level.txt
|
|
13
|
+
test/test_story_test.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
story_test
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
from unittest.mock import Mock
|
|
3
|
+
|
|
4
|
+
from story_test import OllamaRunner, exit_code, run_tests
|
|
5
|
+
|
|
6
|
+
ROOT = Path(__file__).resolve().parent.parent
|
|
7
|
+
TESTS = ROOT / "test/the-tell-tale-heart.tests.yml"
|
|
8
|
+
STORY = ROOT / "test/the-tell-tale-heart.md"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_tell_tale_heart_example() -> None:
|
|
12
|
+
runner = Mock()
|
|
13
|
+
runner.predict.side_effect = [
|
|
14
|
+
{"answers": {"author": {"choice": "true"}}},
|
|
15
|
+
{"answers": {"narrator-reliability": {"choice": "true"}}},
|
|
16
|
+
{"answers": {"victim": {"choice": "true"}}},
|
|
17
|
+
{"answers": {"motive": {"choice": "false"}}},
|
|
18
|
+
{"answers": {"confession": {"choice": "true"}}},
|
|
19
|
+
{"answers": {"location": {"choice": "false"}}},
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
results = run_tests(TESTS, [STORY], runner=runner)
|
|
23
|
+
|
|
24
|
+
assert results == [
|
|
25
|
+
{"name": "author", "passed": True},
|
|
26
|
+
{"name": "narrator-reliability", "passed": True},
|
|
27
|
+
{"name": "victim", "passed": True},
|
|
28
|
+
{"name": "motive", "passed": False},
|
|
29
|
+
{"name": "confession", "passed": True},
|
|
30
|
+
{"name": "location", "passed": False},
|
|
31
|
+
]
|
|
32
|
+
assert runner.predict.call_count == 6
|
|
33
|
+
state, questions = runner.predict.call_args_list[0].args
|
|
34
|
+
assert "The Tell-Tale Heart" in state
|
|
35
|
+
assert state.startswith("Assertion: The story was written by Edgar Allan Poe.")
|
|
36
|
+
assert questions["author"]["type"] == "boolean"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def test_failures_do_not_affect_exit_code_by_default() -> None:
|
|
40
|
+
results = [{"name": "location", "passed": False}]
|
|
41
|
+
|
|
42
|
+
assert exit_code(results, fail_on_test_failure=False) == 0
|
|
43
|
+
assert exit_code(results, fail_on_test_failure=True) == 1
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_ollama_runner_reads_structured_boolean() -> None:
|
|
47
|
+
class Message:
|
|
48
|
+
content = '{"supported": true}'
|
|
49
|
+
|
|
50
|
+
class Response:
|
|
51
|
+
message = Message()
|
|
52
|
+
|
|
53
|
+
class Client:
|
|
54
|
+
def chat(self, **kwargs):
|
|
55
|
+
assert kwargs["model"] == "qwen3:8b"
|
|
56
|
+
assert kwargs["options"]["num_ctx"] == 32768
|
|
57
|
+
return Response()
|
|
58
|
+
|
|
59
|
+
runner = OllamaRunner("qwen3:8b", "http://localhost:11434", 32768)
|
|
60
|
+
runner.client = Client()
|
|
61
|
+
|
|
62
|
+
result = runner.predict("Story text", {"author": {"type": "boolean"}})
|
|
63
|
+
|
|
64
|
+
assert result == {"answers": {"author": {"supported": True}}}
|