researchzosho 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,40 @@
1
+ Metadata-Version: 2.4
2
+ Name: researchzosho
3
+ Version: 0.1.0
4
+ Summary: ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us.
5
+ License: Apache-2.0
6
+ Requires-Python: >=3.9
7
+ Description-Content-Type: text/markdown
8
+
9
+ # researchzosho (Python)
10
+
11
+ **ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
12
+
13
+ A thin client for The Librarian over HTTP — the library protocol, contract 1.0
14
+ (`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
15
+ in the daemon. No dependencies beyond the standard library.
16
+
17
+ ```python
18
+ from researchzosho import Librarian, LibraryError
19
+
20
+ lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
21
+ pkg = lib.ask("How do subtitlers handle keigo?")
22
+ if pkg["holds_nothing"]:
23
+ print("nothing held")
24
+ for e in pkg["entries"]:
25
+ print(e["id"], e["kind"], e["state"], e["title"])
26
+
27
+ v = lib.established("Keigo has no direct English equivalent")
28
+ print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
29
+
30
+ text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
31
+
32
+ job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
33
+ result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
34
+ ```
35
+
36
+ Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
37
+ `.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
38
+ can show to a person.
39
+
40
+ `pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
@@ -0,0 +1,32 @@
1
+ # researchzosho (Python)
2
+
3
+ **ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
4
+
5
+ A thin client for The Librarian over HTTP — the library protocol, contract 1.0
6
+ (`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
7
+ in the daemon. No dependencies beyond the standard library.
8
+
9
+ ```python
10
+ from researchzosho import Librarian, LibraryError
11
+
12
+ lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
13
+ pkg = lib.ask("How do subtitlers handle keigo?")
14
+ if pkg["holds_nothing"]:
15
+ print("nothing held")
16
+ for e in pkg["entries"]:
17
+ print(e["id"], e["kind"], e["state"], e["title"])
18
+
19
+ v = lib.established("Keigo has no direct English equivalent")
20
+ print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
21
+
22
+ text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
23
+
24
+ job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
25
+ result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
26
+ ```
27
+
28
+ Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
29
+ `.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
30
+ can show to a person.
31
+
32
+ `pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
@@ -0,0 +1,15 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "researchzosho"
7
+ version = "0.1.0"
8
+ description = "ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = {text = "Apache-2.0"}
12
+ dependencies = []
13
+
14
+ [tool.setuptools.packages.find]
15
+ include = ["researchzosho*"]
@@ -0,0 +1,197 @@
1
+ """ResearchZosho — client for The Librarian, the library protocol (contract 1.3).
2
+
3
+ Transport and types only. Every method returns the daemon's JSON as a dict; errors raise
4
+ LibraryError with the protocol's stable code.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import time
10
+ import urllib.error
11
+ import urllib.parse
12
+ import urllib.request
13
+ from typing import Any, Dict, List, Optional, Union
14
+
15
+ CONTRACT = "1.0"
16
+
17
+ __all__ = ["Librarian", "LibraryError", "CONTRACT"]
18
+
19
+
20
+ class LibraryError(Exception):
21
+ """A protocol error: .code is stable (not_found, forbidden, no_sources, invalid_args, unavailable)."""
22
+
23
+ def __init__(self, code: str, message: str, status: int = 0):
24
+ super().__init__(message)
25
+ self.code = code
26
+ self.message = message
27
+ self.status = status
28
+
29
+ def __str__(self) -> str: # a sentence a person can read
30
+ return f"{self.message} [{self.code}]"
31
+
32
+
33
+ class Librarian:
34
+ """One library at one URL. Pass the bearer token that proves your did; omit it to be anonymous."""
35
+
36
+ def __init__(self, base_url: str = "http://127.0.0.1:4649", token: Optional[str] = None,
37
+ runtime: str = "python", timeout: float = 60.0):
38
+ self.base_url = base_url.rstrip("/")
39
+ self.token = token
40
+ self.runtime = runtime
41
+ self.timeout = timeout
42
+
43
+ # ---- the nine calls ----
44
+
45
+ def ask(self, question: str, k: int = 6) -> Dict[str, Any]:
46
+ return self._post("ask", {"question": question, "k": k})
47
+
48
+ def search(self, query: str, k: int = 10, subject: Optional[str] = None,
49
+ cursor: Optional[str] = None) -> Dict[str, Any]:
50
+ args: Dict[str, Any] = {"query": query, "k": k}
51
+ if subject:
52
+ args["subject"] = subject
53
+ if cursor:
54
+ args["cursor"] = cursor
55
+ return self._post("search", args)
56
+
57
+ def search_all(self, query: str, subject: Optional[str] = None, page: int = 50) -> List[Dict[str, Any]]:
58
+ """Follow next_cursor to the end."""
59
+ hits: List[Dict[str, Any]] = []
60
+ cursor = None
61
+ while True:
62
+ r = self.search(query, k=page, subject=subject, cursor=cursor)
63
+ hits.extend(r["hits"])
64
+ cursor = r.get("next_cursor")
65
+ if not cursor:
66
+ return hits
67
+
68
+ def get(self, entry_id: str) -> Dict[str, Any]:
69
+ return self._post("get", {"id": entry_id})["entry"]
70
+
71
+ def read(self, locator: str, max_chars: int = 20000) -> Dict[str, Any]:
72
+ return self._post("read", {"locator": locator, "max_chars": max_chars})
73
+
74
+ def established(self, claim: str) -> Dict[str, Any]:
75
+ return self._post("established", {"claim": claim})
76
+
77
+ def submit(self, claim: str, sources: List[Union[str, Dict[str, str]]], claim_type: str = "synthesis",
78
+ confidence: str = "medium", title: Optional[str] = None,
79
+ triple: Optional[Dict[str, str]] = None) -> Dict[str, Any]:
80
+ """triple = {"subject", "predicate", "object"} makes the finding an edge of the graph (library_map)."""
81
+ args: Dict[str, Any] = {"claim": claim, "sources": sources, "claim_type": claim_type, "confidence": confidence}
82
+ if title:
83
+ args["title"] = title
84
+ if triple:
85
+ args["triple"] = triple
86
+ return self._post("submit", args)
87
+
88
+ def frontier(self) -> List[Dict[str, Any]]:
89
+ return self._post("frontier", {"op": "list"})["questions"]
90
+
91
+ def frontier_add(self, question: str) -> Dict[str, Any]:
92
+ return self._post("frontier", {"op": "add", "question": question})
93
+
94
+ def subjects(self) -> List[Dict[str, Any]]:
95
+ return self._post("subjects", {})["subjects"]
96
+
97
+ def status(self) -> Dict[str, Any]:
98
+ return self._post("status", {})
99
+
100
+ def changes(self, since: Union[str, int] = 0, limit: int = 200) -> Dict[str, Any]:
101
+ """Recall notices after a cursor: {changes[], next_cursor, latest, more}. Keep next_cursor between runs."""
102
+ return self._post("changes", {"since": str(since), "limit": limit})
103
+
104
+ # ---- resources ----
105
+
106
+ def resources(self, cursor: Optional[str] = None) -> Dict[str, Any]:
107
+ return self._get("resources", {"cursor": cursor} if cursor else {})
108
+
109
+ def resource(self, uri: str) -> str:
110
+ return self._get("resource", {"uri": uri})["contents"][0]["text"]
111
+
112
+ # ---- research and job are contract 1.2; the jobs page and crews/run are the daemon's own ----
113
+
114
+ def perspectives(self, question: str, max: int = 5) -> Dict[str, Any]:
115
+ """Who studies this and what each would ask: {perspectives[], sub_questions[]} for research()."""
116
+ return self._post("perspectives", {"question": question, "max": max})
117
+
118
+ def map(self, focus: str, depth: int = 1, k: int = 25) -> Dict[str, Any]:
119
+ """The graph around a node name or an entry id: nodes {id, kind, label, also, wikidata}, edges = findings, open questions."""
120
+ return self._post("map", {"focus": focus, "depth": depth, "k": k})
121
+
122
+ def research(self, question: str, mode: str = "broad", max_turns: Optional[int] = None,
123
+ sub_questions: Optional[List[str]] = None, sources: str = "both",
124
+ collections: Optional[List[str]] = None, max_minutes: Optional[int] = None) -> Dict[str, Any]:
125
+ """File an overnight ask. max_turns and max_minutes are ceilings, either, both or neither: the most
126
+ model turns the whole run may spend, the most minutes it may take. With neither the run goes until
127
+ the work is done. sub_questions (≤8) is the plan when you already have one.
128
+ sources: both (shelves first) | shelves (the person's corpus only) | web; collections scope the shelves."""
129
+ body: Dict[str, Any] = {"question": question, "mode": mode, "sources": sources}
130
+ if max_turns:
131
+ body["max_turns"] = int(max_turns)
132
+ if max_minutes:
133
+ body["max_minutes"] = int(max_minutes)
134
+ if sub_questions:
135
+ body["sub_questions"] = list(sub_questions)[:8]
136
+ if collections:
137
+ body["collections"] = list(collections)[:8]
138
+ return self._post("research", body)
139
+
140
+ def jobs(self, limit: int = 20, cursor: Optional[str] = None) -> Dict[str, Any]:
141
+ """A page of this patron's jobs: {active[], finished[] (newest first), next_cursor, finished_total, running, queued}."""
142
+ params: Dict[str, Any] = {"limit": limit}
143
+ if cursor:
144
+ params["cursor"] = cursor
145
+ return self._get("jobs", params)
146
+
147
+ def job(self, job_id: str) -> Dict[str, Any]:
148
+ return self._get(f"jobs/{urllib.parse.quote(job_id, safe='')}", {})
149
+
150
+ def wait(self, job_id: str, poll_s: float = 15.0, timeout_s: Optional[float] = None) -> Dict[str, Any]:
151
+ """Block until the job leaves 'running'. Overnight asks take hours; leave timeout_s None."""
152
+ t0 = time.time()
153
+ while True:
154
+ j = self.job(job_id)
155
+ if j["state"] not in ("running", "queued"):
156
+ return j
157
+ if timeout_s is not None and time.time() - t0 > timeout_s:
158
+ raise TimeoutError(f"job {job_id} still running after {timeout_s}s")
159
+ time.sleep(poll_s)
160
+
161
+ def run_crews(self) -> Dict[str, Any]:
162
+ return self._post("crews/run", {})
163
+
164
+ # ---- plumbing ----
165
+
166
+ def _headers(self) -> Dict[str, str]:
167
+ h = {"Content-Type": "application/json", "Accept": "application/json"}
168
+ if self.token:
169
+ h["Authorization"] = f"Bearer {self.token}"
170
+ return h
171
+
172
+ def _post(self, route: str, args: Dict[str, Any]) -> Dict[str, Any]:
173
+ body = dict(args)
174
+ body["patron"] = {"runtime": self.runtime} # the did comes from the token, never asserted here
175
+ req = urllib.request.Request(f"{self.base_url}/v1/{route}", data=json.dumps(body).encode("utf-8"),
176
+ headers=self._headers(), method="POST")
177
+ return self._send(req)
178
+
179
+ def _get(self, route: str, params: Dict[str, Any]) -> Dict[str, Any]:
180
+ url = f"{self.base_url}/v1/{route}"
181
+ if params:
182
+ url += "?" + urllib.parse.urlencode(params)
183
+ req = urllib.request.Request(url, headers=self._headers(), method="GET")
184
+ return self._send(req)
185
+
186
+ def _send(self, req: urllib.request.Request) -> Dict[str, Any]:
187
+ try:
188
+ with urllib.request.urlopen(req, timeout=self.timeout) as res:
189
+ return json.loads(res.read().decode("utf-8"))
190
+ except urllib.error.HTTPError as e:
191
+ try:
192
+ err = json.loads(e.read().decode("utf-8")).get("error", {})
193
+ except Exception:
194
+ err = {}
195
+ raise LibraryError(err.get("code", "unavailable"), err.get("message", str(e)), e.code) from None
196
+ except urllib.error.URLError as e:
197
+ raise LibraryError("unavailable", f"The Librarian did not answer at {self.base_url}: {e.reason}") from None
@@ -0,0 +1,40 @@
1
+ Metadata-Version: 2.4
2
+ Name: researchzosho
3
+ Version: 0.1.0
4
+ Summary: ResearchZosho client — The Librarian's library protocol (contract 1.3). Research Harness For The Rest Of Us.
5
+ License: Apache-2.0
6
+ Requires-Python: >=3.9
7
+ Description-Content-Type: text/markdown
8
+
9
+ # researchzosho (Python)
10
+
11
+ **ResearchZosho** — 研究蔵書, the research holdings. Research Harness For The Rest Of Us.
12
+
13
+ A thin client for The Librarian over HTTP — the library protocol, contract 1.0
14
+ (`docs/LIBRARY_PROTOCOL.md`). Transport and types only; the library's one implementation lives
15
+ in the daemon. No dependencies beyond the standard library.
16
+
17
+ ```python
18
+ from researchzosho import Librarian, LibraryError
19
+
20
+ lib = Librarian("http://127.0.0.1:4649", token="…") # token from `researchzosho reader token <did>`
21
+ pkg = lib.ask("How do subtitlers handle keigo?")
22
+ if pkg["holds_nothing"]:
23
+ print("nothing held")
24
+ for e in pkg["entries"]:
25
+ print(e["id"], e["kind"], e["state"], e["title"])
26
+
27
+ v = lib.established("Keigo has no direct English equivalent")
28
+ print(v["verdict"], [a["id"] for a in v["accepted"]], [d["id"] for d in v["unreviewed"]])
29
+
30
+ text = lib.read("https://example.org/paper.pdf", max_chars=4000)["text"]
31
+
32
+ job = lib.research("What did Japanese sources say about keigo in subtitles after 2020?")
33
+ result = lib.wait(job["job_id"]) # an overnight ask; poll or wait
34
+ ```
35
+
36
+ Every result carries `library_id`, `library_name`, `contract`. Errors raise `LibraryError` with
37
+ `.code` in `not_found | forbidden | no_sources | invalid_args | unavailable` and a message you
38
+ can show to a person.
39
+
40
+ `pip install researchzosho`. The CLI is `researchzosho` (`zosho` for short).
@@ -0,0 +1,8 @@
1
+ README.md
2
+ pyproject.toml
3
+ researchzosho/__init__.py
4
+ researchzosho.egg-info/PKG-INFO
5
+ researchzosho.egg-info/SOURCES.txt
6
+ researchzosho.egg-info/dependency_links.txt
7
+ researchzosho.egg-info/top_level.txt
8
+ tests/test_live.py
@@ -0,0 +1 @@
1
+ researchzosho
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,31 @@
1
+ """Runs against a live daemon: RESEARCHZOSHO_URL (and optionally RESEARCHZOSHO_TOKEN)."""
2
+ import os
3
+ import pytest
4
+
5
+ from researchzosho import Librarian, LibraryError, CONTRACT
6
+
7
+ URL = os.environ.get("RESEARCHZOSHO_URL")
8
+ TOKEN = os.environ.get("RESEARCHZOSHO_TOKEN")
9
+
10
+ pytestmark = pytest.mark.skipif(not URL, reason="set RESEARCHZOSHO_URL to a running daemon")
11
+
12
+
13
+ def test_status_carries_provenance():
14
+ s = Librarian(URL, TOKEN).status()
15
+ assert s["contract"] == CONTRACT
16
+ assert s["library_id"].startswith("lib_")
17
+ assert "finding" in s["counts"]
18
+
19
+
20
+ def test_errors_carry_codes():
21
+ with pytest.raises(LibraryError) as e:
22
+ Librarian(URL, TOKEN).get("F-0000-nope")
23
+ assert e.value.code == "not_found"
24
+ assert e.value.status == 404
25
+
26
+
27
+ def test_submit_needs_sources_and_proof():
28
+ anon = Librarian(URL)
29
+ with pytest.raises(LibraryError) as e:
30
+ anon.submit("A claim long enough to be refused for lacking sources.", [])
31
+ assert e.value.code in ("forbidden", "no_sources")