researchzosho 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- researchzosho-0.1.0/PKG-INFO +40 -0
- researchzosho-0.1.0/README.md +32 -0
- researchzosho-0.1.0/pyproject.toml +15 -0
- researchzosho-0.1.0/researchzosho/__init__.py +197 -0
- researchzosho-0.1.0/researchzosho.egg-info/PKG-INFO +40 -0
- researchzosho-0.1.0/researchzosho.egg-info/SOURCES.txt +8 -0
- researchzosho-0.1.0/researchzosho.egg-info/dependency_links.txt +1 -0
- researchzosho-0.1.0/researchzosho.egg-info/top_level.txt +1 -0
- researchzosho-0.1.0/setup.cfg +4 -0
- researchzosho-0.1.0/tests/test_live.py +31 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: researchzosho
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.9
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
|
|
9
|
+
# researchzosho (Python)
|
|
10
|
+
|
|
11
|
+
**ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
|
|
12
|
+
|
|
13
|
+
A thin client for The Librarian over HTTP — the library protocol, contract 1.0
|
|
14
|
+
(`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
|
|
15
|
+
in the daemon. No dependencies beyond the standard library.
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from researchzosho import Librarian, LibraryError
|
|
19
|
+
|
|
20
|
+
lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
|
|
21
|
+
pkg = lib.ask("How do subtitlers handle keigo?")
|
|
22
|
+
if pkg["holds_nothing"]:
|
|
23
|
+
print("nothing held")
|
|
24
|
+
for e in pkg["entries"]:
|
|
25
|
+
print(e["id"], e["kind"], e["state"], e["title"])
|
|
26
|
+
|
|
27
|
+
v = lib.established("Keigo has no direct English equivalent")
|
|
28
|
+
print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
|
|
29
|
+
|
|
30
|
+
text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
|
|
31
|
+
|
|
32
|
+
job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
|
|
33
|
+
result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
|
|
37
|
+
`.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
|
|
38
|
+
can show to a person.
|
|
39
|
+
|
|
40
|
+
`pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# researchzosho (Python)
|
|
2
|
+
|
|
3
|
+
**ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
|
|
4
|
+
|
|
5
|
+
A thin client for The Librarian over HTTP — the library protocol, contract 1.0
|
|
6
|
+
(`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
|
|
7
|
+
in the daemon. No dependencies beyond the standard library.
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from researchzosho import Librarian, LibraryError
|
|
11
|
+
|
|
12
|
+
lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
|
|
13
|
+
pkg = lib.ask("How do subtitlers handle keigo?")
|
|
14
|
+
if pkg["holds_nothing"]:
|
|
15
|
+
print("nothing held")
|
|
16
|
+
for e in pkg["entries"]:
|
|
17
|
+
print(e["id"], e["kind"], e["state"], e["title"])
|
|
18
|
+
|
|
19
|
+
v = lib.established("Keigo has no direct English equivalent")
|
|
20
|
+
print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
|
|
21
|
+
|
|
22
|
+
text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
|
|
23
|
+
|
|
24
|
+
job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
|
|
25
|
+
result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
|
|
29
|
+
`.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
|
|
30
|
+
can show to a person.
|
|
31
|
+
|
|
32
|
+
`pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "researchzosho"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = {text = "Apache-2.0"}
|
|
12
|
+
dependencies = []
|
|
13
|
+
|
|
14
|
+
[tool.setuptools.packages.find]
|
|
15
|
+
include = ["researchzosho*"]
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
"""ResearchZosho — client for The Librarian, the library protocol (contract 1.3).
|
|
2
|
+
|
|
3
|
+
Transport and types only. Every method returns the daemon's JSON as a dict; errors raise
|
|
4
|
+
LibraryError with the protocol's stable code.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import time
|
|
10
|
+
import urllib.error
|
|
11
|
+
import urllib.parse
|
|
12
|
+
import urllib.request
|
|
13
|
+
from typing import Any, Dict, List, Optional, Union
|
|
14
|
+
|
|
15
|
+
CONTRACT = "1.0"
|
|
16
|
+
|
|
17
|
+
__all__ = ["Librarian", "LibraryError", "CONTRACT"]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class LibraryError(Exception):
|
|
21
|
+
"""A protocol error: .code is stable (not_found, forbidden, no_sources, invalid_args, unavailable)."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, code: str, message: str, status: int = 0):
|
|
24
|
+
super().__init__(message)
|
|
25
|
+
self.code = code
|
|
26
|
+
self.message = message
|
|
27
|
+
self.status = status
|
|
28
|
+
|
|
29
|
+
def __str__(self) -> str: # a sentence a person can read
|
|
30
|
+
return f"{self.message} [{self.code}]"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Librarian:
|
|
34
|
+
"""One library at one URL. Pass the bearer token that proves your did; omit it to be anonymous."""
|
|
35
|
+
|
|
36
|
+
def __init__(self, base_url: str = "http://127.0.0.1:4649", token: Optional[str] = None,
|
|
37
|
+
runtime: str = "python", timeout: float = 60.0):
|
|
38
|
+
self.base_url = base_url.rstrip("/")
|
|
39
|
+
self.token = token
|
|
40
|
+
self.runtime = runtime
|
|
41
|
+
self.timeout = timeout
|
|
42
|
+
|
|
43
|
+
# ---- the nine calls ----
|
|
44
|
+
|
|
45
|
+
def ask(self, question: str, k: int = 6) -> Dict[str, Any]:
|
|
46
|
+
return self._post("ask", {"question": question, "k": k})
|
|
47
|
+
|
|
48
|
+
def search(self, query: str, k: int = 10, subject: Optional[str] = None,
|
|
49
|
+
cursor: Optional[str] = None) -> Dict[str, Any]:
|
|
50
|
+
args: Dict[str, Any] = {"query": query, "k": k}
|
|
51
|
+
if subject:
|
|
52
|
+
args["subject"] = subject
|
|
53
|
+
if cursor:
|
|
54
|
+
args["cursor"] = cursor
|
|
55
|
+
return self._post("search", args)
|
|
56
|
+
|
|
57
|
+
def search_all(self, query: str, subject: Optional[str] = None, page: int = 50) -> List[Dict[str, Any]]:
|
|
58
|
+
"""Follow next_cursor to the end."""
|
|
59
|
+
hits: List[Dict[str, Any]] = []
|
|
60
|
+
cursor = None
|
|
61
|
+
while True:
|
|
62
|
+
r = self.search(query, k=page, subject=subject, cursor=cursor)
|
|
63
|
+
hits.extend(r["hits"])
|
|
64
|
+
cursor = r.get("next_cursor")
|
|
65
|
+
if not cursor:
|
|
66
|
+
return hits
|
|
67
|
+
|
|
68
|
+
def get(self, entry_id: str) -> Dict[str, Any]:
|
|
69
|
+
return self._post("get", {"id": entry_id})["entry"]
|
|
70
|
+
|
|
71
|
+
def read(self, locator: str, max_chars: int = 20000) -> Dict[str, Any]:
|
|
72
|
+
return self._post("read", {"locator": locator, "max_chars": max_chars})
|
|
73
|
+
|
|
74
|
+
def established(self, claim: str) -> Dict[str, Any]:
|
|
75
|
+
return self._post("established", {"claim": claim})
|
|
76
|
+
|
|
77
|
+
def submit(self, claim: str, sources: List[Union[str, Dict[str, str]]], claim_type: str = "synthesis",
|
|
78
|
+
confidence: str = "medium", title: Optional[str] = None,
|
|
79
|
+
triple: Optional[Dict[str, str]] = None) -> Dict[str, Any]:
|
|
80
|
+
"""triple = {"subject", "predicate", "object"} makes the finding an edge of the graph (library_map)."""
|
|
81
|
+
args: Dict[str, Any] = {"claim": claim, "sources": sources, "claim_type": claim_type, "confidence": confidence}
|
|
82
|
+
if title:
|
|
83
|
+
args["title"] = title
|
|
84
|
+
if triple:
|
|
85
|
+
args["triple"] = triple
|
|
86
|
+
return self._post("submit", args)
|
|
87
|
+
|
|
88
|
+
def frontier(self) -> List[Dict[str, Any]]:
|
|
89
|
+
return self._post("frontier", {"op": "list"})["questions"]
|
|
90
|
+
|
|
91
|
+
def frontier_add(self, question: str) -> Dict[str, Any]:
|
|
92
|
+
return self._post("frontier", {"op": "add", "question": question})
|
|
93
|
+
|
|
94
|
+
def subjects(self) -> List[Dict[str, Any]]:
|
|
95
|
+
return self._post("subjects", {})["subjects"]
|
|
96
|
+
|
|
97
|
+
def status(self) -> Dict[str, Any]:
|
|
98
|
+
return self._post("status", {})
|
|
99
|
+
|
|
100
|
+
def changes(self, since: Union[str, int] = 0, limit: int = 200) -> Dict[str, Any]:
|
|
101
|
+
"""Recall notices after a cursor: {changes[], next_cursor, latest, more}. Keep next_cursor between runs."""
|
|
102
|
+
return self._post("changes", {"since": str(since), "limit": limit})
|
|
103
|
+
|
|
104
|
+
# ---- resources ----
|
|
105
|
+
|
|
106
|
+
def resources(self, cursor: Optional[str] = None) -> Dict[str, Any]:
|
|
107
|
+
return self._get("resources", {"cursor": cursor} if cursor else {})
|
|
108
|
+
|
|
109
|
+
def resource(self, uri: str) -> str:
|
|
110
|
+
return self._get("resource", {"uri": uri})["contents"][0]["text"]
|
|
111
|
+
|
|
112
|
+
# ---- research and job are contract 1.2; the jobs page and crews/run are the daemon's own ----
|
|
113
|
+
|
|
114
|
+
def perspectives(self, question: str, max: int = 5) -> Dict[str, Any]:
|
|
115
|
+
"""Who studies this and what each would ask: {perspectives[], sub_questions[]} for research()."""
|
|
116
|
+
return self._post("perspectives", {"question": question, "max": max})
|
|
117
|
+
|
|
118
|
+
def map(self, focus: str, depth: int = 1, k: int = 25) -> Dict[str, Any]:
|
|
119
|
+
"""The graph around a node name or an entry id: nodes {id, kind, label, also, wikidata}, edges = findings, open questions."""
|
|
120
|
+
return self._post("map", {"focus": focus, "depth": depth, "k": k})
|
|
121
|
+
|
|
122
|
+
def research(self, question: str, mode: str = "broad", max_turns: Optional[int] = None,
|
|
123
|
+
sub_questions: Optional[List[str]] = None, sources: str = "both",
|
|
124
|
+
collections: Optional[List[str]] = None, max_minutes: Optional[int] = None) -> Dict[str, Any]:
|
|
125
|
+
"""File an overnight ask. max_turns and max_minutes are ceilings, either, both or neither: the most
|
|
126
|
+
model turns the whole run may spend, the most minutes it may take. With neither the run goes until
|
|
127
|
+
the work is done. sub_questions (≤8) is the plan when you already have one.
|
|
128
|
+
sources: both (shelves first) | shelves (the person's corpus only) | web; collections scope the shelves."""
|
|
129
|
+
body: Dict[str, Any] = {"question": question, "mode": mode, "sources": sources}
|
|
130
|
+
if max_turns:
|
|
131
|
+
body["max_turns"] = int(max_turns)
|
|
132
|
+
if max_minutes:
|
|
133
|
+
body["max_minutes"] = int(max_minutes)
|
|
134
|
+
if sub_questions:
|
|
135
|
+
body["sub_questions"] = list(sub_questions)[:8]
|
|
136
|
+
if collections:
|
|
137
|
+
body["collections"] = list(collections)[:8]
|
|
138
|
+
return self._post("research", body)
|
|
139
|
+
|
|
140
|
+
def jobs(self, limit: int = 20, cursor: Optional[str] = None) -> Dict[str, Any]:
|
|
141
|
+
"""A page of this patron's jobs: {active[], finished[] (newest first), next_cursor, finished_total, running, queued}."""
|
|
142
|
+
params: Dict[str, Any] = {"limit": limit}
|
|
143
|
+
if cursor:
|
|
144
|
+
params["cursor"] = cursor
|
|
145
|
+
return self._get("jobs", params)
|
|
146
|
+
|
|
147
|
+
def job(self, job_id: str) -> Dict[str, Any]:
|
|
148
|
+
return self._get(f"jobs/{urllib.parse.quote(job_id, safe='')}", {})
|
|
149
|
+
|
|
150
|
+
def wait(self, job_id: str, poll_s: float = 15.0, timeout_s: Optional[float] = None) -> Dict[str, Any]:
|
|
151
|
+
"""Block until the job leaves 'running'. Overnight asks take hours; leave timeout_s None."""
|
|
152
|
+
t0 = time.time()
|
|
153
|
+
while True:
|
|
154
|
+
j = self.job(job_id)
|
|
155
|
+
if j["state"] not in ("running", "queued"):
|
|
156
|
+
return j
|
|
157
|
+
if timeout_s is not None and time.time() - t0 > timeout_s:
|
|
158
|
+
raise TimeoutError(f"job {job_id} still running after {timeout_s}s")
|
|
159
|
+
time.sleep(poll_s)
|
|
160
|
+
|
|
161
|
+
def run_crews(self) -> Dict[str, Any]:
|
|
162
|
+
return self._post("crews/run", {})
|
|
163
|
+
|
|
164
|
+
# ---- plumbing ----
|
|
165
|
+
|
|
166
|
+
def _headers(self) -> Dict[str, str]:
|
|
167
|
+
h = {"Content-Type": "application/json", "Accept": "application/json"}
|
|
168
|
+
if self.token:
|
|
169
|
+
h["Authorization"] = f"Bearer {self.token}"
|
|
170
|
+
return h
|
|
171
|
+
|
|
172
|
+
def _post(self, route: str, args: Dict[str, Any]) -> Dict[str, Any]:
|
|
173
|
+
body = dict(args)
|
|
174
|
+
body["patron"] = {"runtime": self.runtime} # the did comes from the token, never asserted here
|
|
175
|
+
req = urllib.request.Request(f"{self.base_url}/v1/{route}", data=json.dumps(body).encode("utf-8"),
|
|
176
|
+
headers=self._headers(), method="POST")
|
|
177
|
+
return self._send(req)
|
|
178
|
+
|
|
179
|
+
def _get(self, route: str, params: Dict[str, Any]) -> Dict[str, Any]:
|
|
180
|
+
url = f"{self.base_url}/v1/{route}"
|
|
181
|
+
if params:
|
|
182
|
+
url += "?" + urllib.parse.urlencode(params)
|
|
183
|
+
req = urllib.request.Request(url, headers=self._headers(), method="GET")
|
|
184
|
+
return self._send(req)
|
|
185
|
+
|
|
186
|
+
def _send(self, req: urllib.request.Request) -> Dict[str, Any]:
|
|
187
|
+
try:
|
|
188
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as res:
|
|
189
|
+
return json.loads(res.read().decode("utf-8"))
|
|
190
|
+
except urllib.error.HTTPError as e:
|
|
191
|
+
try:
|
|
192
|
+
err = json.loads(e.read().decode("utf-8")).get("error", {})
|
|
193
|
+
except Exception:
|
|
194
|
+
err = {}
|
|
195
|
+
raise LibraryError(err.get("code", "unavailable"), err.get("message", str(e)), e.code) from None
|
|
196
|
+
except urllib.error.URLError as e:
|
|
197
|
+
raise LibraryError("unavailable", f"The Librarian did not answer at {self.base_url}: {e.reason}") from None
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: researchzosho
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.9
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
|
|
9
|
+
# researchzosho (Python)
|
|
10
|
+
|
|
11
|
+
**ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
|
|
12
|
+
|
|
13
|
+
A thin client for The Librarian over HTTP — the library protocol, contract 1.0
|
|
14
|
+
(`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
|
|
15
|
+
in the daemon. No dependencies beyond the standard library.
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from researchzosho import Librarian, LibraryError
|
|
19
|
+
|
|
20
|
+
lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
|
|
21
|
+
pkg = lib.ask("How do subtitlers handle keigo?")
|
|
22
|
+
if pkg["holds_nothing"]:
|
|
23
|
+
print("nothing held")
|
|
24
|
+
for e in pkg["entries"]:
|
|
25
|
+
print(e["id"], e["kind"], e["state"], e["title"])
|
|
26
|
+
|
|
27
|
+
v = lib.established("Keigo has no direct English equivalent")
|
|
28
|
+
print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
|
|
29
|
+
|
|
30
|
+
text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
|
|
31
|
+
|
|
32
|
+
job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
|
|
33
|
+
result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
|
|
37
|
+
`.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
|
|
38
|
+
can show to a person.
|
|
39
|
+
|
|
40
|
+
`pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
researchzosho
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Runs against a live daemon: RESEARCHZOSHO_URL (and optionally RESEARCHZOSHO_TOKEN)."""
|
|
2
|
+
import os
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from researchzosho import Librarian, LibraryError, CONTRACT
|
|
6
|
+
|
|
7
|
+
URL = os.environ.get("RESEARCHZOSHO_URL")
|
|
8
|
+
TOKEN = os.environ.get("RESEARCHZOSHO_TOKEN")
|
|
9
|
+
|
|
10
|
+
pytestmark = pytest.mark.skipif(not URL, reason="set RESEARCHZOSHO_URL to a running daemon")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def test_status_carries_provenance():
|
|
14
|
+
s = Librarian(URL, TOKEN).status()
|
|
15
|
+
assert s["contract"] == CONTRACT
|
|
16
|
+
assert s["library_id"].startswith("lib_")
|
|
17
|
+
assert "finding" in s["counts"]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_errors_carry_codes():
|
|
21
|
+
with pytest.raises(LibraryError) as e:
|
|
22
|
+
Librarian(URL, TOKEN).get("F-0000-nope")
|
|
23
|
+
assert e.value.code == "not_found"
|
|
24
|
+
assert e.value.status == 404
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_submit_needs_sources_and_proof():
|
|
28
|
+
anon = Librarian(URL)
|
|
29
|
+
with pytest.raises(LibraryError) as e:
|
|
30
|
+
anon.submit("A claim long enough to be refused for lacking sources.", [])
|
|
31
|
+
assert e.value.code in ("forbidden", "no_sources")
|