lyrenth-research 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lyrenth_research-0.1.0/.gitignore +7 -0
- lyrenth_research-0.1.0/LICENSE +21 -0
- lyrenth_research-0.1.0/PKG-INFO +181 -0
- lyrenth_research-0.1.0/README.md +163 -0
- lyrenth_research-0.1.0/pyproject.toml +30 -0
- lyrenth_research-0.1.0/src/lyrenth_research/__init__.py +30 -0
- lyrenth_research-0.1.0/src/lyrenth_research/answer.py +111 -0
- lyrenth_research-0.1.0/src/lyrenth_research/cli.py +111 -0
- lyrenth_research-0.1.0/src/lyrenth_research/sources.py +125 -0
- lyrenth_research-0.1.0/tests/test_research.py +236 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Aleksma AI, Inc.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: lyrenth-research
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A research agent that reads its sources: give it a question and URLs, get an answer with numbered citations.
|
|
5
|
+
Project-URL: Homepage, https://lyrenth.com
|
|
6
|
+
Project-URL: Documentation, https://lyrenth.com/docs/quickstart
|
|
7
|
+
Author: Lyrenth
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: agent,aidocument,citations,llm,lyrenth,rag,research,web
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Requires-Dist: lyrenth>=0.1.1
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# lyrenth-research
|
|
20
|
+
|
|
21
|
+
A research agent that actually reads its sources.
|
|
22
|
+
|
|
23
|
+
Every AI answer that reads the web pays for reading pages, and Lyrenth
|
|
24
|
+
makes that reading several times cheaper. This agent is the easiest way to
|
|
25
|
+
see it.
|
|
26
|
+
|
|
27
|
+
Give it a question and a list of URLs. It reads every page through
|
|
28
|
+
[Lyrenth](https://lyrenth.com) as clean text, drops duplicates, stays inside
|
|
29
|
+
a token budget, tells you which pages it could not read, and asks your model
|
|
30
|
+
for an answer with numbered citations. Then it checks that every citation
|
|
31
|
+
points at a source it really read.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install lyrenth-research
|
|
35
|
+
export LYRENTH_API_KEY=... # free key: https://lyrenth.com/signup
|
|
36
|
+
export LLM_BASE_URL=http://localhost:11434/v1 # any OpenAI-compatible endpoint
|
|
37
|
+
export LLM_MODEL=your-model-name
|
|
38
|
+
|
|
39
|
+
lyrenth-research "What is the difference between a web crawler and web indexing?" \
|
|
40
|
+
https://en.wikipedia.org/wiki/Web_indexing \
|
|
41
|
+
https://en.wikipedia.org/wiki/Web_crawler
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Why
|
|
45
|
+
|
|
46
|
+
Most research agents get the finding and the writing right and treat the
|
|
47
|
+
reading as one line: fetch the page, strip some tags, paste it into the
|
|
48
|
+
prompt. That line decides more about the answer than the prompt does. An
|
|
49
|
+
agent that reads badly pays for navigation menus and cookie banners, reads
|
|
50
|
+
the same article twice under two URLs, cites pages it never really read,
|
|
51
|
+
and quietly answers from nothing when a source fails.
|
|
52
|
+
|
|
53
|
+
This agent does five things on purpose:
|
|
54
|
+
|
|
55
|
+
1. **Reads clean text, not HTML.** Every page arrives as an AIDocument:
|
|
56
|
+
Markdown content, the canonical URL, when it was read, and how many
|
|
57
|
+
tokens it costs.
|
|
58
|
+
2. **Keeps provenance.** Each source keeps its title, canonical URL and
|
|
59
|
+
read time, and all three go into the prompt.
|
|
60
|
+
3. **Reads each document once.** A mobile copy, a tracking parameter or an
|
|
61
|
+
old redirect resolves to the same canonical URL and is dropped.
|
|
62
|
+
4. **Stays inside a budget.** Sources are added in the order you give them
|
|
63
|
+
until the next one would exceed the budget. Put the ones you trust most
|
|
64
|
+
first.
|
|
65
|
+
5. **Fails visibly.** Every page that was skipped is listed with the reason,
|
|
66
|
+
and a citation to a source number that does not exist is flagged.
|
|
67
|
+
|
|
68
|
+
## A real run
|
|
69
|
+
|
|
70
|
+
Reading step only (`--sources-only`), against the live API on September 19,
|
|
71
|
+
2026, unedited. The progress report goes to stderr, the numbered context to
|
|
72
|
+
stdout (here, `context.md`).
|
|
73
|
+
|
|
74
|
+
```text
|
|
75
|
+
$ lyrenth-research "What is the difference between a web crawler and web indexing?" \
|
|
76
|
+
https://en.wikipedia.org/wiki/Web_indexing \
|
|
77
|
+
https://en.wikipedia.org/wiki/Web_crawler \
|
|
78
|
+
https://en.wikipedia.org/wiki/Robots.txt \
|
|
79
|
+
https://en.wikipedia.org/wiki/No_such_page_for_this_example \
|
|
80
|
+
https://en.m.wikipedia.org/wiki/Web_indexing \
|
|
81
|
+
--sources-only > context.md
|
|
82
|
+
skipped https://en.wikipedia.org/wiki/Robots.txt: over budget (16,719 tokens would exceed 30,000)
|
|
83
|
+
skipped https://en.wikipedia.org/wiki/No_such_page_for_this_example: origin returned 404
|
|
84
|
+
skipped https://en.m.wikipedia.org/wiki/Web_indexing: duplicate of https://en.wikipedia.org/wiki/Web_indexing
|
|
85
|
+
read [1] Web indexing - Wikipedia 3,119 tokens https://en.wikipedia.org/wiki/Web_indexing
|
|
86
|
+
read [2] Web crawler - Wikipedia 22,410 tokens https://en.wikipedia.org/wiki/Web_crawler
|
|
87
|
+
context 25,529 tokens from 2 sources (raw HTML would be 111,672)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The same question with a model configured, run on September 19, 2026 with
|
|
91
|
+
a hosted model through its OpenAI-compatible endpoint. The reading report is
|
|
92
|
+
identical; this is the end of the answer and the source list, shortened for
|
|
93
|
+
length and otherwise as printed:
|
|
94
|
+
|
|
95
|
+
```text
|
|
96
|
+
**Key Difference:**
|
|
97
|
+
- The web crawler is the tool or agent that collects web content by navigating the web.
|
|
98
|
+
- Web indexing is the process that takes the content gathered by the crawler and organizes it into a searchable index.
|
|
99
|
+
|
|
100
|
+
In summary:
|
|
101
|
+
**Web crawling** is about collecting web pages, while **web indexing** is about organizing and making sense of the collected data for efficient search and retrieval[1][2].
|
|
102
|
+
|
|
103
|
+
Sources
|
|
104
|
+
[1] Web indexing - Wikipedia https://en.wikipedia.org/wiki/Web_indexing
|
|
105
|
+
[2] Web crawler - Wikipedia https://en.wikipedia.org/wiki/Web_crawler
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Every claim carries a source number, both sources were cited, and neither
|
|
109
|
+
is marked `(not cited)`.
|
|
110
|
+
|
|
111
|
+
## Two keys, and why
|
|
112
|
+
|
|
113
|
+
- **`LYRENTH_API_KEY`** reads the pages. The free tier includes 2,000 reads
|
|
114
|
+
a month; one question with five sources uses five.
|
|
115
|
+
- **Your model** writes the answer. Anything that speaks the
|
|
116
|
+
OpenAI-compatible chat completions protocol works: the large hosted
|
|
117
|
+
providers, most gateways, and local servers such as Ollama or llama.cpp.
|
|
118
|
+
Set `LLM_BASE_URL` and `LLM_MODEL`, plus `LLM_API_KEY` if your endpoint
|
|
119
|
+
needs one. Nothing is sent anywhere else.
|
|
120
|
+
|
|
121
|
+
Only need the reading? `--sources-only` prints the numbered, cited context
|
|
122
|
+
without calling a model, ready to drop into your own pipeline. Add `--json`
|
|
123
|
+
for structured output.
|
|
124
|
+
|
|
125
|
+
## Options
|
|
126
|
+
|
|
127
|
+
```text
|
|
128
|
+
lyrenth-research QUESTION [URL ...] [options]
|
|
129
|
+
|
|
130
|
+
-f, --file FILE more URLs, one per line (lines starting with # are ignored)
|
|
131
|
+
--budget N token budget for all sources together (default 30000)
|
|
132
|
+
--fresh ask for a fresh fetch instead of the stored page
|
|
133
|
+
--sources-only read and print the context, no model call
|
|
134
|
+
--json JSON output
|
|
135
|
+
--model-url URL OpenAI-compatible base URL (default: LLM_BASE_URL)
|
|
136
|
+
--model NAME model name (default: LLM_MODEL)
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## As a library
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from lyrenth_research import gather, answer
|
|
143
|
+
|
|
144
|
+
gathered = gather(
|
|
145
|
+
["https://en.wikipedia.org/wiki/Web_indexing",
|
|
146
|
+
"https://en.wikipedia.org/wiki/Web_crawler"],
|
|
147
|
+
token_budget=30_000,
|
|
148
|
+
)
|
|
149
|
+
for skipped in gathered.skipped:
|
|
150
|
+
print("skipped", skipped.url, skipped.reason)
|
|
151
|
+
|
|
152
|
+
result = answer("How do crawlers and indexes relate?", gathered.sources)
|
|
153
|
+
print(result.text)
|
|
154
|
+
print("cited:", result.cited, "unknown:", result.unknown_citations)
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
`gather` returns the sources it read, the ones it skipped with a reason, and
|
|
158
|
+
the tokens used. `answer` returns the model's text, the source numbers it
|
|
159
|
+
cited, and any cited numbers that do not exist.
|
|
160
|
+
|
|
161
|
+
## Where the URLs come from
|
|
162
|
+
|
|
163
|
+
On purpose, not from here. Your agent may get them from a search provider,
|
|
164
|
+
from the user, from links inside pages it already read, or from a curated
|
|
165
|
+
list of trusted sites. Finding and reading are separate jobs; keeping them
|
|
166
|
+
separate means you can change one without breaking the other.
|
|
167
|
+
|
|
168
|
+
## Tests
|
|
169
|
+
|
|
170
|
+
```bash
|
|
171
|
+
pip install -e .
|
|
172
|
+
python tests/test_research.py
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
No network and no keys needed: reading is tested against a fake client, and
|
|
176
|
+
the answer step against a local HTTP server that speaks the chat
|
|
177
|
+
completions protocol.
|
|
178
|
+
|
|
179
|
+
## License
|
|
180
|
+
|
|
181
|
+
MIT
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# lyrenth-research
|
|
2
|
+
|
|
3
|
+
A research agent that actually reads its sources.
|
|
4
|
+
|
|
5
|
+
Every AI answer that reads the web pays for reading pages, and Lyrenth
|
|
6
|
+
makes that reading several times cheaper. This agent is the easiest way to
|
|
7
|
+
see it.
|
|
8
|
+
|
|
9
|
+
Give it a question and a list of URLs. It reads every page through
|
|
10
|
+
[Lyrenth](https://lyrenth.com) as clean text, drops duplicates, stays inside
|
|
11
|
+
a token budget, tells you which pages it could not read, and asks your model
|
|
12
|
+
for an answer with numbered citations. Then it checks that every citation
|
|
13
|
+
points at a source it really read.
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install lyrenth-research
|
|
17
|
+
export LYRENTH_API_KEY=... # free key: https://lyrenth.com/signup
|
|
18
|
+
export LLM_BASE_URL=http://localhost:11434/v1 # any OpenAI-compatible endpoint
|
|
19
|
+
export LLM_MODEL=your-model-name
|
|
20
|
+
|
|
21
|
+
lyrenth-research "What is the difference between a web crawler and web indexing?" \
|
|
22
|
+
https://en.wikipedia.org/wiki/Web_indexing \
|
|
23
|
+
https://en.wikipedia.org/wiki/Web_crawler
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Why
|
|
27
|
+
|
|
28
|
+
Most research agents get the finding and the writing right and treat the
|
|
29
|
+
reading as one line: fetch the page, strip some tags, paste it into the
|
|
30
|
+
prompt. That line decides more about the answer than the prompt does. An
|
|
31
|
+
agent that reads badly pays for navigation menus and cookie banners, reads
|
|
32
|
+
the same article twice under two URLs, cites pages it never really read,
|
|
33
|
+
and quietly answers from nothing when a source fails.
|
|
34
|
+
|
|
35
|
+
This agent does five things on purpose:
|
|
36
|
+
|
|
37
|
+
1. **Reads clean text, not HTML.** Every page arrives as an AIDocument:
|
|
38
|
+
Markdown content, the canonical URL, when it was read, and how many
|
|
39
|
+
tokens it costs.
|
|
40
|
+
2. **Keeps provenance.** Each source keeps its title, canonical URL and
|
|
41
|
+
read time, and all three go into the prompt.
|
|
42
|
+
3. **Reads each document once.** A mobile copy, a tracking parameter or an
|
|
43
|
+
old redirect resolves to the same canonical URL and is dropped.
|
|
44
|
+
4. **Stays inside a budget.** Sources are added in the order you give them
|
|
45
|
+
until the next one would exceed the budget. Put the ones you trust most
|
|
46
|
+
first.
|
|
47
|
+
5. **Fails visibly.** Every page that was skipped is listed with the reason,
|
|
48
|
+
and a citation to a source number that does not exist is flagged.
|
|
49
|
+
|
|
50
|
+
## A real run
|
|
51
|
+
|
|
52
|
+
Reading step only (`--sources-only`), against the live API on September 19,
|
|
53
|
+
2026, unedited. The progress report goes to stderr, the numbered context to
|
|
54
|
+
stdout (here, `context.md`).
|
|
55
|
+
|
|
56
|
+
```text
|
|
57
|
+
$ lyrenth-research "What is the difference between a web crawler and web indexing?" \
|
|
58
|
+
https://en.wikipedia.org/wiki/Web_indexing \
|
|
59
|
+
https://en.wikipedia.org/wiki/Web_crawler \
|
|
60
|
+
https://en.wikipedia.org/wiki/Robots.txt \
|
|
61
|
+
https://en.wikipedia.org/wiki/No_such_page_for_this_example \
|
|
62
|
+
https://en.m.wikipedia.org/wiki/Web_indexing \
|
|
63
|
+
--sources-only > context.md
|
|
64
|
+
skipped https://en.wikipedia.org/wiki/Robots.txt: over budget (16,719 tokens would exceed 30,000)
|
|
65
|
+
skipped https://en.wikipedia.org/wiki/No_such_page_for_this_example: origin returned 404
|
|
66
|
+
skipped https://en.m.wikipedia.org/wiki/Web_indexing: duplicate of https://en.wikipedia.org/wiki/Web_indexing
|
|
67
|
+
read [1] Web indexing - Wikipedia 3,119 tokens https://en.wikipedia.org/wiki/Web_indexing
|
|
68
|
+
read [2] Web crawler - Wikipedia 22,410 tokens https://en.wikipedia.org/wiki/Web_crawler
|
|
69
|
+
context 25,529 tokens from 2 sources (raw HTML would be 111,672)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The same question with a model configured, run on September 19, 2026 with
|
|
73
|
+
a hosted model through its OpenAI-compatible endpoint. The reading report is
|
|
74
|
+
identical; this is the end of the answer and the source list, shortened for
|
|
75
|
+
length and otherwise as printed:
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
**Key Difference:**
|
|
79
|
+
- The web crawler is the tool or agent that collects web content by navigating the web.
|
|
80
|
+
- Web indexing is the process that takes the content gathered by the crawler and organizes it into a searchable index.
|
|
81
|
+
|
|
82
|
+
In summary:
|
|
83
|
+
**Web crawling** is about collecting web pages, while **web indexing** is about organizing and making sense of the collected data for efficient search and retrieval[1][2].
|
|
84
|
+
|
|
85
|
+
Sources
|
|
86
|
+
[1] Web indexing - Wikipedia https://en.wikipedia.org/wiki/Web_indexing
|
|
87
|
+
[2] Web crawler - Wikipedia https://en.wikipedia.org/wiki/Web_crawler
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Every claim carries a source number, both sources were cited, and neither
|
|
91
|
+
is marked `(not cited)`.
|
|
92
|
+
|
|
93
|
+
## Two keys, and why
|
|
94
|
+
|
|
95
|
+
- **`LYRENTH_API_KEY`** reads the pages. The free tier includes 2,000 reads
|
|
96
|
+
a month; one question with five sources uses five.
|
|
97
|
+
- **Your model** writes the answer. Anything that speaks the
|
|
98
|
+
OpenAI-compatible chat completions protocol works: the large hosted
|
|
99
|
+
providers, most gateways, and local servers such as Ollama or llama.cpp.
|
|
100
|
+
Set `LLM_BASE_URL` and `LLM_MODEL`, plus `LLM_API_KEY` if your endpoint
|
|
101
|
+
needs one. Nothing is sent anywhere else.
|
|
102
|
+
|
|
103
|
+
Only need the reading? `--sources-only` prints the numbered, cited context
|
|
104
|
+
without calling a model, ready to drop into your own pipeline. Add `--json`
|
|
105
|
+
for structured output.
|
|
106
|
+
|
|
107
|
+
## Options
|
|
108
|
+
|
|
109
|
+
```text
|
|
110
|
+
lyrenth-research QUESTION [URL ...] [options]
|
|
111
|
+
|
|
112
|
+
-f, --file FILE more URLs, one per line (lines starting with # are ignored)
|
|
113
|
+
--budget N token budget for all sources together (default 30000)
|
|
114
|
+
--fresh ask for a fresh fetch instead of the stored page
|
|
115
|
+
--sources-only read and print the context, no model call
|
|
116
|
+
--json JSON output
|
|
117
|
+
--model-url URL OpenAI-compatible base URL (default: LLM_BASE_URL)
|
|
118
|
+
--model NAME model name (default: LLM_MODEL)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
## As a library
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
from lyrenth_research import gather, answer
|
|
125
|
+
|
|
126
|
+
gathered = gather(
|
|
127
|
+
["https://en.wikipedia.org/wiki/Web_indexing",
|
|
128
|
+
"https://en.wikipedia.org/wiki/Web_crawler"],
|
|
129
|
+
token_budget=30_000,
|
|
130
|
+
)
|
|
131
|
+
for skipped in gathered.skipped:
|
|
132
|
+
print("skipped", skipped.url, skipped.reason)
|
|
133
|
+
|
|
134
|
+
result = answer("How do crawlers and indexes relate?", gathered.sources)
|
|
135
|
+
print(result.text)
|
|
136
|
+
print("cited:", result.cited, "unknown:", result.unknown_citations)
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
`gather` returns the sources it read, the ones it skipped with a reason, and
|
|
140
|
+
the tokens used. `answer` returns the model's text, the source numbers it
|
|
141
|
+
cited, and any cited numbers that do not exist.
|
|
142
|
+
|
|
143
|
+
## Where the URLs come from
|
|
144
|
+
|
|
145
|
+
On purpose, not from here. Your agent may get them from a search provider,
|
|
146
|
+
from the user, from links inside pages it already read, or from a curated
|
|
147
|
+
list of trusted sites. Finding and reading are separate jobs; keeping them
|
|
148
|
+
separate means you can change one without breaking the other.
|
|
149
|
+
|
|
150
|
+
## Tests
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
pip install -e .
|
|
154
|
+
python tests/test_research.py
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
No network and no keys needed: reading is tested against a fake client, and
|
|
158
|
+
the answer step against a local HTTP server that speaks the chat
|
|
159
|
+
completions protocol.
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
MIT
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "lyrenth-research"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A research agent that reads its sources: give it a question and URLs, get an answer with numbered citations."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [{ name = "Lyrenth" }]
|
|
13
|
+
keywords = ["research", "agent", "citations", "rag", "llm", "web", "lyrenth", "aidocument"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
"Environment :: Console",
|
|
19
|
+
]
|
|
20
|
+
dependencies = ["lyrenth>=0.1.1"]
|
|
21
|
+
|
|
22
|
+
[project.scripts]
|
|
23
|
+
lyrenth-research = "lyrenth_research.cli:main"
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://lyrenth.com"
|
|
27
|
+
Documentation = "https://lyrenth.com/docs/quickstart"
|
|
28
|
+
|
|
29
|
+
[tool.hatch.build.targets.wheel]
|
|
30
|
+
packages = ["src/lyrenth_research"]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""A research agent that reads its sources.
|
|
2
|
+
|
|
3
|
+
from lyrenth_research import gather, answer
|
|
4
|
+
|
|
5
|
+
gathered = gather(["https://example.com/a", "https://example.com/b"])
|
|
6
|
+
result = answer("What does the page say about X?", gathered.sources)
|
|
7
|
+
print(result.text)
|
|
8
|
+
|
|
9
|
+
Reading goes through Lyrenth (LYRENTH_API_KEY, free key at
|
|
10
|
+
https://lyrenth.com/signup). The answer comes from any OpenAI-compatible
|
|
11
|
+
model endpoint you configure with LLM_BASE_URL and LLM_MODEL.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
__version__ = "0.1.0"
|
|
15
|
+
|
|
16
|
+
from .answer import Answer, ModelError, answer, build_context, check_citations # noqa: E402
|
|
17
|
+
from .sources import Gathered, Skipped, Source, gather # noqa: E402
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Answer",
|
|
21
|
+
"Gathered",
|
|
22
|
+
"ModelError",
|
|
23
|
+
"Skipped",
|
|
24
|
+
"Source",
|
|
25
|
+
"answer",
|
|
26
|
+
"build_context",
|
|
27
|
+
"check_citations",
|
|
28
|
+
"gather",
|
|
29
|
+
"__version__",
|
|
30
|
+
]
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Turning sources into an answer with numbered citations.
|
|
2
|
+
|
|
3
|
+
The model is yours. Anything that speaks the OpenAI-compatible chat
|
|
4
|
+
completions protocol works: the large hosted providers, most gateways, and
|
|
5
|
+
local servers such as Ollama or llama.cpp. Configure it with three
|
|
6
|
+
environment variables (or the matching CLI flags):
|
|
7
|
+
|
|
8
|
+
LLM_BASE_URL e.g. http://localhost:11434/v1 for a local Ollama
|
|
9
|
+
LLM_MODEL the model name your endpoint expects
|
|
10
|
+
LLM_API_KEY optional; local servers usually need none
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.request
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from typing import List, Optional
|
|
22
|
+
|
|
23
|
+
from .sources import Source
|
|
24
|
+
|
|
25
|
+
SYSTEM_PROMPT = (
|
|
26
|
+
"You answer questions using only the numbered sources you are given. "
|
|
27
|
+
"Cite every claim with the number of the source it comes from, like [1] or [2][3]. "
|
|
28
|
+
"If the sources do not contain the answer, say so plainly instead of guessing. "
|
|
29
|
+
"Do not use knowledge that is not in the sources."
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class ModelError(RuntimeError):
|
|
34
|
+
"""The model endpoint was not configured, unreachable, or answered badly."""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class Answer:
|
|
39
|
+
text: str
|
|
40
|
+
sources: List[Source]
|
|
41
|
+
cited: List[int] = field(default_factory=list)
|
|
42
|
+
unknown_citations: List[int] = field(default_factory=list)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def build_context(sources: List[Source]) -> str:
|
|
46
|
+
"""Number the sources and keep where and when each was read."""
|
|
47
|
+
parts = []
|
|
48
|
+
for i, s in enumerate(sources, 1):
|
|
49
|
+
header = f"[{i}] {s.title or s.url}\nURL: {s.url}"
|
|
50
|
+
if s.fetched_at:
|
|
51
|
+
header += f"\nRead: {s.fetched_at}"
|
|
52
|
+
parts.append(f"{header}\n\n{s.text}")
|
|
53
|
+
return "\n\n---\n\n".join(parts)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def build_messages(question: str, sources: List[Source]) -> list:
|
|
57
|
+
user = f"Question: {question}\n\nSources:\n\n{build_context(sources)}"
|
|
58
|
+
return [
|
|
59
|
+
{"role": "system", "content": SYSTEM_PROMPT},
|
|
60
|
+
{"role": "user", "content": user},
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def check_citations(text: str, n_sources: int):
|
|
65
|
+
"""Return (cited, unknown): source numbers used, and numbers that do not exist."""
|
|
66
|
+
found = sorted({int(m) for m in re.findall(r"\[(\d+)\]", text)})
|
|
67
|
+
cited = [n for n in found if 1 <= n <= n_sources]
|
|
68
|
+
unknown = [n for n in found if n < 1 or n > n_sources]
|
|
69
|
+
return cited, unknown
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def call_model(
|
|
73
|
+
messages: list,
|
|
74
|
+
base_url: Optional[str] = None,
|
|
75
|
+
model: Optional[str] = None,
|
|
76
|
+
api_key: Optional[str] = None,
|
|
77
|
+
timeout: float = 180.0,
|
|
78
|
+
) -> str:
|
|
79
|
+
base_url = (base_url or os.environ.get("LLM_BASE_URL", "")).rstrip("/")
|
|
80
|
+
model = model or os.environ.get("LLM_MODEL", "")
|
|
81
|
+
api_key = api_key if api_key is not None else os.environ.get("LLM_API_KEY", "")
|
|
82
|
+
if not base_url or not model:
|
|
83
|
+
raise ModelError(
|
|
84
|
+
"no model configured: set LLM_BASE_URL and LLM_MODEL "
|
|
85
|
+
"(or pass --model-url and --model), or use --sources-only"
|
|
86
|
+
)
|
|
87
|
+
headers = {"Content-Type": "application/json"}
|
|
88
|
+
if api_key:
|
|
89
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
90
|
+
body = json.dumps({"model": model, "messages": messages, "temperature": 0}).encode("utf-8")
|
|
91
|
+
req = urllib.request.Request(f"{base_url}/chat/completions", data=body, headers=headers, method="POST")
|
|
92
|
+
try:
|
|
93
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
94
|
+
payload = json.loads(resp.read().decode("utf-8"))
|
|
95
|
+
except urllib.error.HTTPError as e:
|
|
96
|
+
detail = e.read().decode("utf-8", "replace")[:300]
|
|
97
|
+
raise ModelError(f"model endpoint returned HTTP {e.code}: {detail}") from None
|
|
98
|
+
except urllib.error.URLError as e:
|
|
99
|
+
raise ModelError(f"could not reach the model endpoint: {e.reason}") from None
|
|
100
|
+
try:
|
|
101
|
+
return payload["choices"][0]["message"]["content"].strip()
|
|
102
|
+
except (KeyError, IndexError, TypeError, AttributeError):
|
|
103
|
+
raise ModelError("model endpoint answered in an unexpected shape") from None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def answer(question: str, sources: List[Source], **model_kwargs) -> Answer:
|
|
107
|
+
if not sources:
|
|
108
|
+
return Answer(text="None of the sources could be read, so there is nothing to answer from.", sources=[])
|
|
109
|
+
text = call_model(build_messages(question, sources), **model_kwargs)
|
|
110
|
+
cited, unknown = check_citations(text, len(sources))
|
|
111
|
+
return Answer(text=text, sources=sources, cited=cited, unknown_citations=unknown)
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Command line: lyrenth-research "question" URL [URL ...]"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
from dataclasses import asdict
|
|
9
|
+
|
|
10
|
+
from lyrenth import Lyrenth, LyrenthError
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .answer import ModelError, answer, build_context
|
|
14
|
+
from .sources import gather
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _parser() -> argparse.ArgumentParser:
|
|
18
|
+
p = argparse.ArgumentParser(
|
|
19
|
+
prog="lyrenth-research",
|
|
20
|
+
description="Read the given URLs through Lyrenth and answer the question with numbered citations.",
|
|
21
|
+
)
|
|
22
|
+
p.add_argument("question", help="the question to answer")
|
|
23
|
+
p.add_argument("urls", nargs="*", help="source URLs, most trusted first")
|
|
24
|
+
p.add_argument("-f", "--file", help="read more URLs from a file, one per line")
|
|
25
|
+
p.add_argument("--budget", type=int, default=30_000, help="token budget for all sources together (default 30000)")
|
|
26
|
+
p.add_argument("--fresh", action="store_true", help="ask Lyrenth for a fresh fetch instead of the stored page")
|
|
27
|
+
p.add_argument("--sources-only", action="store_true", help="read the sources and print the cited context, without calling a model")
|
|
28
|
+
p.add_argument("--json", action="store_true", help="print the result as JSON")
|
|
29
|
+
p.add_argument("--model-url", help="OpenAI-compatible base URL (default: LLM_BASE_URL)")
|
|
30
|
+
p.add_argument("--model", help="model name (default: LLM_MODEL)")
|
|
31
|
+
p.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
32
|
+
return p
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _urls(args) -> list:
|
|
36
|
+
urls = list(args.urls)
|
|
37
|
+
if args.file:
|
|
38
|
+
with open(args.file, encoding="utf-8") as fh:
|
|
39
|
+
urls += [line.strip() for line in fh if line.strip() and not line.startswith("#")]
|
|
40
|
+
return urls
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _report(gathered, out=None) -> None:
|
|
44
|
+
# Looked up at call time, so a caller that redirects stderr captures it.
|
|
45
|
+
out = out or sys.stderr
|
|
46
|
+
for s in gathered.skipped:
|
|
47
|
+
print(f"skipped {s.url}: {s.reason}", file=out)
|
|
48
|
+
for i, s in enumerate(gathered.sources, 1):
|
|
49
|
+
print(f"read [{i}] {s.title} {s.tokens:,} tokens {s.url}", file=out)
|
|
50
|
+
raw = gathered.raw_html_tokens
|
|
51
|
+
line = f"context {gathered.tokens:,} tokens from {len(gathered.sources)} sources"
|
|
52
|
+
if raw:
|
|
53
|
+
line += f" (raw HTML would be {raw:,})"
|
|
54
|
+
print(line, file=out)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def main(argv=None) -> int:
|
|
58
|
+
args = _parser().parse_args(argv)
|
|
59
|
+
urls = _urls(args)
|
|
60
|
+
if not urls:
|
|
61
|
+
print("give at least one URL, or a file of URLs with --file", file=sys.stderr)
|
|
62
|
+
return 2
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
gathered = gather(urls, client=Lyrenth(), token_budget=args.budget, fresh=args.fresh)
|
|
66
|
+
except LyrenthError as e:
|
|
67
|
+
print(f"error: {e}", file=sys.stderr)
|
|
68
|
+
return 1
|
|
69
|
+
|
|
70
|
+
_report(gathered)
|
|
71
|
+
|
|
72
|
+
if args.sources_only:
|
|
73
|
+
if args.json:
|
|
74
|
+
print(json.dumps({"sources": [asdict(s) for s in gathered.sources],
|
|
75
|
+
"skipped": [asdict(s) for s in gathered.skipped]}, indent=2))
|
|
76
|
+
else:
|
|
77
|
+
print(build_context(gathered.sources))
|
|
78
|
+
return 0
|
|
79
|
+
|
|
80
|
+
try:
|
|
81
|
+
result = answer(args.question, gathered.sources, base_url=args.model_url, model=args.model)
|
|
82
|
+
except ModelError as e:
|
|
83
|
+
print(f"error: {e}", file=sys.stderr)
|
|
84
|
+
return 1
|
|
85
|
+
|
|
86
|
+
if args.json:
|
|
87
|
+
print(json.dumps({
|
|
88
|
+
"question": args.question,
|
|
89
|
+
"answer": result.text,
|
|
90
|
+
"cited": result.cited,
|
|
91
|
+
"unknown_citations": result.unknown_citations,
|
|
92
|
+
"sources": [{"n": i, "title": s.title, "url": s.url, "read": s.fetched_at}
|
|
93
|
+
for i, s in enumerate(result.sources, 1)],
|
|
94
|
+
"skipped": [asdict(s) for s in gathered.skipped],
|
|
95
|
+
}, indent=2))
|
|
96
|
+
return 0
|
|
97
|
+
|
|
98
|
+
print(result.text)
|
|
99
|
+
print()
|
|
100
|
+
print("Sources")
|
|
101
|
+
for i, s in enumerate(result.sources, 1):
|
|
102
|
+
mark = "" if i in result.cited else " (not cited)"
|
|
103
|
+
print(f" [{i}] {s.title} {s.url}{mark}")
|
|
104
|
+
if result.unknown_citations:
|
|
105
|
+
nums = ", ".join(f"[{n}]" for n in result.unknown_citations)
|
|
106
|
+
print(f"\nwarning: the answer cites {nums}, which is not one of the sources above", file=sys.stderr)
|
|
107
|
+
return 0
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
if __name__ == "__main__":
|
|
111
|
+
sys.exit(main())
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Reading sources properly: once each, with provenance, inside a budget.
|
|
2
|
+
|
|
3
|
+
This is the part of a research agent that decides what the model gets to
|
|
4
|
+
see. It reads every URL through Lyrenth, which returns each page as a clean
|
|
5
|
+
AIDocument with its canonical URL and a measured token count, and then:
|
|
6
|
+
|
|
7
|
+
- drops a page that was already read under another URL (mobile copies,
|
|
8
|
+
tracking parameters, redirects all resolve to one canonical URL);
|
|
9
|
+
- stops adding sources once the token budget would be exceeded, so one
|
|
10
|
+
long page cannot crowd out several short ones;
|
|
11
|
+
- records every page it could not read, with the reason, instead of
|
|
12
|
+
quietly leaving a gap.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from typing import Iterable, List, Optional
|
|
19
|
+
|
|
20
|
+
from lyrenth import Lyrenth
|
|
21
|
+
|
|
22
|
+
# The batch endpoint takes up to 20 URLs per call.
|
|
23
|
+
BATCH_SIZE = 20
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class Source:
|
|
28
|
+
"""One page the agent read, with everything a citation needs."""
|
|
29
|
+
|
|
30
|
+
url: str
|
|
31
|
+
title: str
|
|
32
|
+
text: str
|
|
33
|
+
fetched_at: str = ""
|
|
34
|
+
tokens: int = 0
|
|
35
|
+
raw_html_tokens: int = 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Skipped:
|
|
40
|
+
"""A URL that did not make it into the sources, and why."""
|
|
41
|
+
|
|
42
|
+
url: str
|
|
43
|
+
reason: str
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class Gathered:
|
|
48
|
+
sources: List[Source] = field(default_factory=list)
|
|
49
|
+
skipped: List[Skipped] = field(default_factory=list)
|
|
50
|
+
tokens: int = 0
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def raw_html_tokens(self) -> int:
|
|
54
|
+
return sum(s.raw_html_tokens for s in self.sources)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _estimate_tokens(text: str) -> int:
|
|
58
|
+
# Only used when a response carries no economics block. Four characters
|
|
59
|
+
# per token is the usual rough figure for English text.
|
|
60
|
+
return max(1, len(text) // 4)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _to_source(doc) -> Source:
|
|
64
|
+
raw = doc.raw or {}
|
|
65
|
+
economics = raw.get("economics") or {}
|
|
66
|
+
source = raw.get("source") or {}
|
|
67
|
+
tokens = int(economics.get("output_tokens_approx") or 0) or _estimate_tokens(doc.markdown)
|
|
68
|
+
return Source(
|
|
69
|
+
url=doc.url,
|
|
70
|
+
title=doc.title,
|
|
71
|
+
text=doc.markdown,
|
|
72
|
+
fetched_at=source.get("fetched_at") or "",
|
|
73
|
+
tokens=tokens,
|
|
74
|
+
raw_html_tokens=int(economics.get("raw_html_tokens_approx") or 0),
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _dedupe_input(urls: Iterable[str]) -> List[str]:
|
|
79
|
+
seen, out = set(), []
|
|
80
|
+
for u in urls:
|
|
81
|
+
u = u.strip()
|
|
82
|
+
if u and u not in seen:
|
|
83
|
+
seen.add(u)
|
|
84
|
+
out.append(u)
|
|
85
|
+
return out
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def gather(
|
|
89
|
+
urls: Iterable[str],
|
|
90
|
+
client: Optional[Lyrenth] = None,
|
|
91
|
+
token_budget: int = 30_000,
|
|
92
|
+
fresh: bool = False,
|
|
93
|
+
) -> Gathered:
|
|
94
|
+
"""Read `urls` in order and return the sources that fit the budget.
|
|
95
|
+
|
|
96
|
+
Order matters: earlier URLs are preferred when the budget runs out, so
|
|
97
|
+
put the sources you trust most first.
|
|
98
|
+
"""
|
|
99
|
+
client = client or Lyrenth()
|
|
100
|
+
urls = _dedupe_input(urls)
|
|
101
|
+
out = Gathered()
|
|
102
|
+
seen_canonical = set()
|
|
103
|
+
|
|
104
|
+
for start in range(0, len(urls), BATCH_SIZE):
|
|
105
|
+
batch = urls[start : start + BATCH_SIZE]
|
|
106
|
+
for result in client.read_batch(batch, fresh=fresh):
|
|
107
|
+
if not result.ok or result.document is None:
|
|
108
|
+
out.skipped.append(Skipped(result.url, result.error or "could not be read"))
|
|
109
|
+
continue
|
|
110
|
+
src = _to_source(result.document)
|
|
111
|
+
if not src.text.strip():
|
|
112
|
+
out.skipped.append(Skipped(result.url, "page has no readable text"))
|
|
113
|
+
continue
|
|
114
|
+
if src.url in seen_canonical:
|
|
115
|
+
out.skipped.append(Skipped(result.url, f"duplicate of {src.url}"))
|
|
116
|
+
continue
|
|
117
|
+
if out.tokens + src.tokens > token_budget:
|
|
118
|
+
out.skipped.append(
|
|
119
|
+
Skipped(result.url, f"over budget ({src.tokens:,} tokens would exceed {token_budget:,})")
|
|
120
|
+
)
|
|
121
|
+
continue
|
|
122
|
+
seen_canonical.add(src.url)
|
|
123
|
+
out.sources.append(src)
|
|
124
|
+
out.tokens += src.tokens
|
|
125
|
+
return out
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Tests for lyrenth-research. No network and no keys needed.
|
|
2
|
+
|
|
3
|
+
python tests/test_research.py # plain, no pytest needed
|
|
4
|
+
pytest tests/ # if pytest is installed
|
|
5
|
+
|
|
6
|
+
Reading is tested against a fake Lyrenth client. The answer step is tested
|
|
7
|
+
against a real HTTP server on localhost that speaks the OpenAI-compatible
|
|
8
|
+
chat completions protocol, so the request the agent sends is checked for real.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import io
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import sys
|
|
15
|
+
import threading
|
|
16
|
+
from contextlib import redirect_stderr, redirect_stdout
|
|
17
|
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
18
|
+
|
|
19
|
+
HERE = os.path.dirname(__file__)
|
|
20
|
+
sys.path.insert(0, os.path.join(HERE, "..", "src"))
|
|
21
|
+
# Prefer the SDK from this monorepo when it is next door, else the installed one.
|
|
22
|
+
sdk = os.path.join(HERE, "..", "..", "lyrenth-python", "src")
|
|
23
|
+
if os.path.isdir(sdk):
|
|
24
|
+
sys.path.insert(0, sdk)
|
|
25
|
+
|
|
26
|
+
from lyrenth import AIDocument, BatchResult # noqa: E402
|
|
27
|
+
|
|
28
|
+
import lyrenth_research.cli as cli # noqa: E402
|
|
29
|
+
from lyrenth_research import ( # noqa: E402
|
|
30
|
+
ModelError,
|
|
31
|
+
Source,
|
|
32
|
+
answer,
|
|
33
|
+
build_context,
|
|
34
|
+
check_citations,
|
|
35
|
+
gather,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def doc(canonical, title, text, tokens, raw=0, fetched="2026-09-19T12:00:00Z"):
|
|
40
|
+
return AIDocument(
|
|
41
|
+
url=canonical,
|
|
42
|
+
title=title,
|
|
43
|
+
markdown=text,
|
|
44
|
+
raw={
|
|
45
|
+
"source": {"url": canonical, "canonical_url": canonical, "fetched_at": fetched},
|
|
46
|
+
"economics": {"output_tokens_approx": tokens, "raw_html_tokens_approx": raw},
|
|
47
|
+
},
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class FakeClient:
|
|
52
|
+
"""Answers read_batch from a dict of url -> AIDocument or error string."""
|
|
53
|
+
|
|
54
|
+
def __init__(self, pages):
|
|
55
|
+
self.pages = pages
|
|
56
|
+
self.calls = []
|
|
57
|
+
|
|
58
|
+
def read_batch(self, urls, fresh=False, max_tokens=None):
|
|
59
|
+
self.calls.append(list(urls))
|
|
60
|
+
out = []
|
|
61
|
+
for u in urls:
|
|
62
|
+
v = self.pages.get(u, "upstream_not_found")
|
|
63
|
+
if isinstance(v, str):
|
|
64
|
+
out.append(BatchResult(url=u, ok=False, error=v))
|
|
65
|
+
else:
|
|
66
|
+
out.append(BatchResult(url=u, ok=True, document=v))
|
|
67
|
+
return out
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
A = "https://example.com/a"
|
|
71
|
+
A_MOBILE = "https://m.example.com/a"
|
|
72
|
+
B = "https://example.com/b"
|
|
73
|
+
C = "https://example.com/c"
|
|
74
|
+
MISSING = "https://example.com/missing"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def pages():
|
|
78
|
+
return {
|
|
79
|
+
A: doc(A, "Page A", "Alpha text.", 3000, 25000),
|
|
80
|
+
A_MOBILE: doc(A, "Page A", "Alpha text.", 3000, 25000),
|
|
81
|
+
B: doc(B, "Page B", "Beta text.", 20000, 80000),
|
|
82
|
+
C: doc(C, "Page C", "Gamma text.", 16000, 40000),
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# ---------------------------------------------------------------- gather
|
|
87
|
+
|
|
88
|
+
def test_gather_every_branch():
|
|
89
|
+
g = gather([A, B, C, MISSING, A_MOBILE, A], client=FakeClient(pages()), token_budget=30000)
|
|
90
|
+
assert [s.url for s in g.sources] == [A, B], g.sources
|
|
91
|
+
assert g.tokens == 23000
|
|
92
|
+
assert g.raw_html_tokens == 105000
|
|
93
|
+
reasons = {s.url: s.reason for s in g.skipped}
|
|
94
|
+
assert "over budget" in reasons[C], reasons
|
|
95
|
+
assert reasons[MISSING] == "upstream_not_found"
|
|
96
|
+
assert reasons[A_MOBILE] == f"duplicate of {A}"
|
|
97
|
+
# The exact same URL given twice is read once, not reported as skipped.
|
|
98
|
+
assert len(g.skipped) == 3
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def test_gather_prefers_earlier_urls_when_budget_is_tight():
|
|
102
|
+
g = gather([C, A, B], client=FakeClient(pages()), token_budget=20000)
|
|
103
|
+
assert [s.url for s in g.sources] == [C, A]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_gather_batches_by_twenty():
|
|
107
|
+
many = {f"https://example.com/{i}": doc(f"https://example.com/{i}", str(i), "x", 10) for i in range(45)}
|
|
108
|
+
client = FakeClient(many)
|
|
109
|
+
g = gather(list(many), client=client, token_budget=10**6)
|
|
110
|
+
assert [len(c) for c in client.calls] == [20, 20, 5]
|
|
111
|
+
assert len(g.sources) == 45
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def test_gather_skips_empty_pages():
|
|
115
|
+
p = {A: doc(A, "Empty", " \n", 5)}
|
|
116
|
+
g = gather([A], client=FakeClient(p))
|
|
117
|
+
assert g.sources == [] and g.skipped[0].reason == "page has no readable text"
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def test_gather_estimates_tokens_without_economics():
|
|
121
|
+
d = AIDocument(url=A, title="A", markdown="x" * 400, raw={"source": {"url": A}})
|
|
122
|
+
g = gather([A], client=FakeClient({A: d}))
|
|
123
|
+
assert g.sources[0].tokens == 100
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ---------------------------------------------------------------- answer helpers
|
|
127
|
+
|
|
128
|
+
def test_build_context_numbers_and_keeps_provenance():
|
|
129
|
+
ctx = build_context([Source(A, "Page A", "Alpha.", "2026-09-19T12:00:00Z"), Source(B, "", "Beta.")])
|
|
130
|
+
assert ctx.startswith("[1] Page A\nURL: https://example.com/a\nRead: 2026-09-19T12:00:00Z\n\nAlpha.")
|
|
131
|
+
assert "[2] https://example.com/b\nURL: https://example.com/b\n\nBeta." in ctx
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def test_check_citations_flags_numbers_that_do_not_exist():
|
|
135
|
+
cited, unknown = check_citations("Alpha [1]. Beta [2][2]. Nonsense [7]. Zero [0].", 2)
|
|
136
|
+
assert cited == [1, 2] and unknown == [0, 7]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def test_answer_without_sources_does_not_call_a_model():
|
|
140
|
+
r = answer("q", [], base_url="http://127.0.0.1:9", model="m")
|
|
141
|
+
assert r.sources == [] and "nothing to answer from" in r.text
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def test_missing_model_config_is_a_clear_error():
|
|
145
|
+
saved = {k: os.environ.pop(k, None) for k in ("LLM_BASE_URL", "LLM_MODEL")}
|
|
146
|
+
try:
|
|
147
|
+
answer("q", [Source(A, "A", "Alpha.")])
|
|
148
|
+
except ModelError as e:
|
|
149
|
+
assert "--sources-only" in str(e)
|
|
150
|
+
else:
|
|
151
|
+
raise AssertionError("expected ModelError")
|
|
152
|
+
finally:
|
|
153
|
+
for k, v in saved.items():
|
|
154
|
+
if v is not None:
|
|
155
|
+
os.environ[k] = v
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# ---------------------------------------------------------------- a real HTTP model endpoint
|
|
159
|
+
|
|
160
|
+
class _Model(BaseHTTPRequestHandler):
|
|
161
|
+
received = []
|
|
162
|
+
reply = "Alpha says hello [1]. Beta agrees [2]."
|
|
163
|
+
|
|
164
|
+
def do_POST(self):
|
|
165
|
+
body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
|
166
|
+
_Model.received.append({"path": self.path, "auth": self.headers.get("Authorization"), "body": body})
|
|
167
|
+
out = json.dumps({"choices": [{"message": {"role": "assistant", "content": _Model.reply}}]}).encode()
|
|
168
|
+
self.send_response(200)
|
|
169
|
+
self.send_header("Content-Type", "application/json")
|
|
170
|
+
self.send_header("Content-Length", str(len(out)))
|
|
171
|
+
self.end_headers()
|
|
172
|
+
self.wfile.write(out)
|
|
173
|
+
|
|
174
|
+
def log_message(self, *a):
|
|
175
|
+
pass
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _serve():
|
|
179
|
+
srv = HTTPServer(("127.0.0.1", 0), _Model)
|
|
180
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
181
|
+
return srv, f"http://127.0.0.1:{srv.server_address[1]}/v1"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def test_answer_calls_an_openai_compatible_endpoint():
|
|
185
|
+
srv, url = _serve()
|
|
186
|
+
try:
|
|
187
|
+
_Model.received.clear()
|
|
188
|
+
_Model.reply = "Alpha says hello [1]. Beta agrees [2]. Also [5]."
|
|
189
|
+
r = answer("What do they say?", [Source(A, "Page A", "Alpha."), Source(B, "Page B", "Beta.")],
|
|
190
|
+
base_url=url, model="test-model", api_key="secret")
|
|
191
|
+
sent = _Model.received[0]
|
|
192
|
+
assert sent["path"] == "/v1/chat/completions"
|
|
193
|
+
assert sent["auth"] == "Bearer secret"
|
|
194
|
+
assert sent["body"]["model"] == "test-model" and sent["body"]["temperature"] == 0
|
|
195
|
+
user = sent["body"]["messages"][1]["content"]
|
|
196
|
+
assert user.startswith("Question: What do they say?") and "[2] Page B" in user
|
|
197
|
+
assert r.cited == [1, 2] and r.unknown_citations == [5]
|
|
198
|
+
finally:
|
|
199
|
+
srv.shutdown()
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def test_cli_end_to_end_with_fake_reader_and_local_model():
|
|
203
|
+
srv, url = _serve()
|
|
204
|
+
real = cli.Lyrenth
|
|
205
|
+
cli.Lyrenth = lambda: FakeClient(pages())
|
|
206
|
+
try:
|
|
207
|
+
_Model.reply = "Alpha is first [1]."
|
|
208
|
+
out, err = io.StringIO(), io.StringIO()
|
|
209
|
+
with redirect_stdout(out), redirect_stderr(err):
|
|
210
|
+
code = cli.main(["What is first?", A, MISSING, B, "--model-url", url, "--model", "m"])
|
|
211
|
+
assert code == 0, err.getvalue()
|
|
212
|
+
assert "skipped https://example.com/missing: upstream_not_found" in err.getvalue()
|
|
213
|
+
assert "Alpha is first [1]." in out.getvalue()
|
|
214
|
+
assert "[2] Page B https://example.com/b (not cited)" in out.getvalue()
|
|
215
|
+
|
|
216
|
+
out = io.StringIO()
|
|
217
|
+
with redirect_stdout(out), redirect_stderr(io.StringIO()):
|
|
218
|
+
assert cli.main(["q", A, B, "--sources-only", "--json"]) == 0
|
|
219
|
+
payload = json.loads(out.getvalue())
|
|
220
|
+
assert [s["url"] for s in payload["sources"]] == [A, B]
|
|
221
|
+
finally:
|
|
222
|
+
cli.Lyrenth = real
|
|
223
|
+
srv.shutdown()
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def test_cli_needs_a_url():
|
|
227
|
+
with redirect_stderr(io.StringIO()):
|
|
228
|
+
assert cli.main(["question only"]) == 2
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
if __name__ == "__main__":
|
|
232
|
+
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
|
|
233
|
+
for t in tests:
|
|
234
|
+
t()
|
|
235
|
+
print(f"ok {t.__name__}")
|
|
236
|
+
print(f"{len(tests)} passed")
|