knos 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- knos/__init__.py +3 -0
- knos/answer.py +368 -0
- knos/cli.py +426 -0
- knos/code.py +271 -0
- knos/errors.py +118 -0
- knos/git.py +97 -0
- knos/help.py +117 -0
- knos/link.py +113 -0
- knos/mcp.py +387 -0
- knos/memory.py +520 -0
- knos/paths.py +46 -0
- knos/pay.py +106 -0
- knos/private.py +133 -0
- knos/sessions.py +269 -0
- knos/team.py +207 -0
- knos-0.1.0.dist-info/METADATA +331 -0
- knos-0.1.0.dist-info/RECORD +20 -0
- knos-0.1.0.dist-info/WHEEL +4 -0
- knos-0.1.0.dist-info/entry_points.txt +2 -0
- knos-0.1.0.dist-info/licenses/LICENSE +21 -0
knos/__init__.py
ADDED
knos/answer.py
ADDED
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
"""Reading a repo, and answering from what was read.
|
|
2
|
+
|
|
3
|
+
knos has no model and no API key. It does not write prose. An answer is the
|
|
4
|
+
passages it found and, beside each one, exactly where it came from: a file
|
|
5
|
+
and line, a session and date, or a commit. A developer will not trust it
|
|
6
|
+
twice without that.
|
|
7
|
+
|
|
8
|
+
Everything here runs because a person ran a command.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from . import code, errors, git, private
|
|
19
|
+
from .memory import INTERNAL, PERSON, Fact, Memory
|
|
20
|
+
|
|
21
|
+
STOP = {
|
|
22
|
+
"a", "about", "an", "and", "are", "as", "at", "be", "but", "by", "did",
|
|
23
|
+
"do", "does", "for", "from", "had", "has", "have", "how", "in", "is",
|
|
24
|
+
"it", "last", "of", "on", "or", "our", "that", "the", "their", "then",
|
|
25
|
+
"there", "this", "to", "was", "we", "were", "what", "when", "where",
|
|
26
|
+
"which", "who", "why", "with", "you", "your",
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Passage:
|
|
32
|
+
"""One thing knos found, and where it came from."""
|
|
33
|
+
|
|
34
|
+
text: str
|
|
35
|
+
source: str # "session" | "git" | "code" | "note"
|
|
36
|
+
where: str
|
|
37
|
+
score: float = 0.0
|
|
38
|
+
path: str = ""
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def line(self) -> str:
|
|
42
|
+
return f"{self.text}\n {self.where}"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# knos is not an archive. It keeps enough of a passage to recognise and
|
|
46
|
+
# search, and names where the whole thing still lives, so a person can go
|
|
47
|
+
# and read it in full. Storing less of each passage is always better than
|
|
48
|
+
# storing fewer of them: a repo whose oldest half is missing has holes a
|
|
49
|
+
# person cannot see, while shorter passages still point at everything.
|
|
50
|
+
KEEP_CHARS = 280
|
|
51
|
+
|
|
52
|
+
# How many good passages count as an answer. Below this, a prose question
|
|
53
|
+
# also asks the code reader, and pays the seconds that costs.
|
|
54
|
+
ENOUGH = 4
|
|
55
|
+
|
|
56
|
+
# What a deliberately written note is worth over derived chatter. Somebody
|
|
57
|
+
# chose to write it down; a session that happens to say the word twenty times
|
|
58
|
+
# did not. Without this the standing rule loses to the argument about it.
|
|
59
|
+
NOTE_LEAD = 1.0
|
|
60
|
+
|
|
61
|
+
# What a code result is worth over prose when the question was structural.
|
|
62
|
+
# Enough to outrank any single passage, so the file and line come first.
|
|
63
|
+
STRUCTURAL_LEAD = 2.0
|
|
64
|
+
|
|
65
|
+
# Ways of asking about the shape of the code rather than the story behind it.
|
|
66
|
+
# These always reach the code reader, however much prose was found, because
|
|
67
|
+
# no amount of session chatter answers "who calls this".
|
|
68
|
+
STRUCTURAL_PHRASES = (
|
|
69
|
+
"who calls",
|
|
70
|
+
"what calls",
|
|
71
|
+
"who uses",
|
|
72
|
+
"what uses",
|
|
73
|
+
"where is",
|
|
74
|
+
"where are",
|
|
75
|
+
"defined",
|
|
76
|
+
"definition",
|
|
77
|
+
"implemented",
|
|
78
|
+
"implementation",
|
|
79
|
+
"function",
|
|
80
|
+
"class",
|
|
81
|
+
"method",
|
|
82
|
+
"call graph",
|
|
83
|
+
"depends on",
|
|
84
|
+
"imports",
|
|
85
|
+
"signature",
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# A name a programmer would type: CamelCase, snake_case, a call, or a file.
|
|
89
|
+
SYMBOL = re.compile(
|
|
90
|
+
r"[a-z]+_[a-z_]+|[a-z]+[A-Z]\w*|[A-Z][a-z]+[A-Z]\w*|\w+\(\)|[\w/\\.-]+\.[a-z]{1,4}\b"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def looks_structural(question: str) -> bool:
|
|
95
|
+
"""Whether this asks about the shape of the code.
|
|
96
|
+
|
|
97
|
+
Asked of the question rather than of the results, because a question
|
|
98
|
+
about a symbol deserves the code reader even when the sessions happen to
|
|
99
|
+
be full of chatter that mentions it.
|
|
100
|
+
"""
|
|
101
|
+
low = question.lower()
|
|
102
|
+
if any(phrase in low for phrase in STRUCTURAL_PHRASES):
|
|
103
|
+
return True
|
|
104
|
+
return bool(SYMBOL.search(question))
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def topic_of(fact: str) -> str:
|
|
108
|
+
"""What to file a fact under when nobody said.
|
|
109
|
+
|
|
110
|
+
The words that carry it, so "always use pnpm, never npm" files under
|
|
111
|
+
"pnpm npm" and is found by asking about either. Asking a person to
|
|
112
|
+
invent a filing name before they can write one sentence down is the
|
|
113
|
+
friction this exists to avoid.
|
|
114
|
+
"""
|
|
115
|
+
# A standing instruction is mostly instruction. What it is *about* is
|
|
116
|
+
# what is left once the telling-off is removed.
|
|
117
|
+
telling = {
|
|
118
|
+
"always", "never", "must", "should", "dont", "please", "make",
|
|
119
|
+
"sure", "keep", "stop", "avoid", "prefer", "only", "ever", "when",
|
|
120
|
+
"use", "using", "used", "into", "onto", "over", "under", "before",
|
|
121
|
+
"after", "instead", "rather", "than", "them", "they",
|
|
122
|
+
}
|
|
123
|
+
# Not a length filter: npm, git, ssh and api are all three letters and
|
|
124
|
+
# all exactly what a rule is about.
|
|
125
|
+
words = [w for w in terms(fact) if w not in telling][:3]
|
|
126
|
+
return " ".join(words) or fact[:40].strip()
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _trim(text: str) -> str:
|
|
130
|
+
text = " ".join(text.split())
|
|
131
|
+
if len(text) <= KEEP_CHARS:
|
|
132
|
+
return text
|
|
133
|
+
return text[:KEEP_CHARS].rsplit(" ", 1)[0] + "..."
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# ---- reading a repo ---------------------------------------------------
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def point(
|
|
140
|
+
repo: Path,
|
|
141
|
+
mem: Memory,
|
|
142
|
+
index_code: bool = True,
|
|
143
|
+
on_progress: Any = None,
|
|
144
|
+
) -> dict[str, int]:
|
|
145
|
+
"""Read this repo: its sessions, its commits, its structure.
|
|
146
|
+
|
|
147
|
+
Called by `knos point`. Facts are stated as they were found; nothing is
|
|
148
|
+
summarised, scored or inferred.
|
|
149
|
+
"""
|
|
150
|
+
from . import sessions
|
|
151
|
+
|
|
152
|
+
counts: dict = {
|
|
153
|
+
"sessions": 0,
|
|
154
|
+
"commits": 0,
|
|
155
|
+
"code": 0,
|
|
156
|
+
"private": 0,
|
|
157
|
+
"full": 0,
|
|
158
|
+
"skipped": [],
|
|
159
|
+
}
|
|
160
|
+
say = on_progress or (lambda *_: None)
|
|
161
|
+
repo = Path(repo).resolve()
|
|
162
|
+
|
|
163
|
+
mem.set_reference(INTERNAL + "repo", {"path": str(repo), "name": repo.name})
|
|
164
|
+
|
|
165
|
+
# Newest first, so that if the store fills up what knos kept is the part
|
|
166
|
+
# anyone is likely to ask about.
|
|
167
|
+
say("looking for past agent sessions")
|
|
168
|
+
for turn in reversed(sessions.read_all(repo)):
|
|
169
|
+
if not mem.record(
|
|
170
|
+
Fact(
|
|
171
|
+
text=_trim(turn.text),
|
|
172
|
+
source="session",
|
|
173
|
+
where=turn.where,
|
|
174
|
+
when=turn.when,
|
|
175
|
+
about=turn.client,
|
|
176
|
+
)
|
|
177
|
+
):
|
|
178
|
+
counts["full"] = 1
|
|
179
|
+
break
|
|
180
|
+
counts["sessions"] += 1
|
|
181
|
+
if counts["sessions"] % 100 == 0:
|
|
182
|
+
say(f"{counts['sessions']} things said in past sessions")
|
|
183
|
+
|
|
184
|
+
if not counts["full"]:
|
|
185
|
+
# Commits arrive newest first, so the first time a file or a person
|
|
186
|
+
# is seen is their latest. A canonical record is written once, then
|
|
187
|
+
# left alone: rewriting it for every older commit produced the same
|
|
188
|
+
# answer after tens of thousands of pointless writes, which was
|
|
189
|
+
# most of what `knos point` spent its time doing.
|
|
190
|
+
named: set[str] = set()
|
|
191
|
+
say("reading the commits")
|
|
192
|
+
for commit in git.read_commits(repo):
|
|
193
|
+
visible = [f for f in commit.files if not private.is_private(repo, f)]
|
|
194
|
+
counts["private"] += len(commit.files) - len(visible)
|
|
195
|
+
if not mem.record(
|
|
196
|
+
Fact(
|
|
197
|
+
text=_trim(commit.text),
|
|
198
|
+
source="git",
|
|
199
|
+
where=commit.where,
|
|
200
|
+
when=commit.when,
|
|
201
|
+
about=commit.author,
|
|
202
|
+
path=visible[0] if visible else "",
|
|
203
|
+
)
|
|
204
|
+
):
|
|
205
|
+
counts["full"] = 1
|
|
206
|
+
break
|
|
207
|
+
# Only people get a canonical record here. A file used to get one
|
|
208
|
+
# per commit that touched it, which on a real repo was hundreds of
|
|
209
|
+
# writes at twenty-five milliseconds each, for something the
|
|
210
|
+
# journal already says: the commit that changed a file names the
|
|
211
|
+
# file. That was most of what reading a repo cost.
|
|
212
|
+
if commit.author not in named:
|
|
213
|
+
named.add(commit.author)
|
|
214
|
+
mem.note_thing(
|
|
215
|
+
PERSON,
|
|
216
|
+
commit.author,
|
|
217
|
+
{"last_commit": commit.short, "last_seen": commit.when[:10]},
|
|
218
|
+
)
|
|
219
|
+
counts["commits"] += 1
|
|
220
|
+
if counts["commits"] % 100 == 0:
|
|
221
|
+
say(f"{counts['commits']} commits")
|
|
222
|
+
|
|
223
|
+
if index_code and code.installed():
|
|
224
|
+
result: dict = {}
|
|
225
|
+
# The longest stretch, and the one with nothing to count as it
|
|
226
|
+
# goes, so say plainly that a wait here is expected.
|
|
227
|
+
say("reading the code, the slow part")
|
|
228
|
+
try:
|
|
229
|
+
result = code.index(repo)
|
|
230
|
+
except Exception:
|
|
231
|
+
# Structure is one source of three. Losing it is worth a line,
|
|
232
|
+
# not the whole command.
|
|
233
|
+
counts["skipped"].append(
|
|
234
|
+
errors.unreadable("code structure", "the reader would not start")
|
|
235
|
+
)
|
|
236
|
+
counts["code"] = int(result.get("nodes") or 0)
|
|
237
|
+
mem.set_reference(INTERNAL + "code_index", result)
|
|
238
|
+
|
|
239
|
+
# What goes into the store has to be plain data, so the skipped files are
|
|
240
|
+
# counted here and handed back to the caller in full.
|
|
241
|
+
mem.set_focus(
|
|
242
|
+
{
|
|
243
|
+
"repo": str(repo),
|
|
244
|
+
"read": {k: v for k, v in counts.items() if isinstance(v, int)},
|
|
245
|
+
"skipped": len(counts["skipped"]),
|
|
246
|
+
}
|
|
247
|
+
)
|
|
248
|
+
return counts
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
# ---- answering --------------------------------------------------------
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def terms(question: str) -> list[str]:
|
|
255
|
+
words = re.findall(r"[A-Za-z_][A-Za-z0-9_.\-]{1,}", question.lower())
|
|
256
|
+
return [w for w in words if w not in STOP and len(w) > 2]
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _score(text: str, wanted: list[str]) -> float:
|
|
260
|
+
low = text.lower()
|
|
261
|
+
hits = sum(1 for w in wanted if w in low)
|
|
262
|
+
if not hits:
|
|
263
|
+
return 0.0
|
|
264
|
+
# Prefer passages that cover more of the question over ones that repeat
|
|
265
|
+
# a single word, and prefer shorter passages at equal coverage.
|
|
266
|
+
return hits / len(wanted) + min(len(low), 2000) / 200000.0
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def ask(
|
|
270
|
+
repo: Path,
|
|
271
|
+
mem: Memory,
|
|
272
|
+
question: str,
|
|
273
|
+
identity: str = private.OWNER,
|
|
274
|
+
limit: int = 8,
|
|
275
|
+
allowed: list[str] | None = None,
|
|
276
|
+
) -> list[Passage]:
|
|
277
|
+
"""Ranked passages with their sources. No prose, no synthesis.
|
|
278
|
+
|
|
279
|
+
For an agent, private paths are dropped before ranking and are never
|
|
280
|
+
counted, so the reply carries no sign that anything was withheld.
|
|
281
|
+
"""
|
|
282
|
+
wanted = terms(question)
|
|
283
|
+
if not wanted:
|
|
284
|
+
return []
|
|
285
|
+
|
|
286
|
+
found: list[Passage] = []
|
|
287
|
+
|
|
288
|
+
# The store's full-text search wants terms, not a sentence: asking it for
|
|
289
|
+
# the whole question makes every word mandatory and finds nothing. Each
|
|
290
|
+
# term is asked for separately and the results are ranked together, so a
|
|
291
|
+
# passage that covers more of the question wins.
|
|
292
|
+
for term in wanted[:6]:
|
|
293
|
+
for hit in mem.search(term, limit=60):
|
|
294
|
+
text = str(hit.get("text") or "").strip()
|
|
295
|
+
if not text:
|
|
296
|
+
continue
|
|
297
|
+
# A canonical record says what knos knows *about* a thing, not
|
|
298
|
+
# where it heard it, so it has no source to name and no business
|
|
299
|
+
# being quoted as an answer. The journal carries the same fact
|
|
300
|
+
# with the commit it came from. `about` is where these belong.
|
|
301
|
+
if hit.get("tier") == "entity":
|
|
302
|
+
continue
|
|
303
|
+
# A note that has been forgotten stops being an answer. The
|
|
304
|
+
# journal keeps the line, because the journal is a record of what
|
|
305
|
+
# was learned and when, but a forgotten thing is not something
|
|
306
|
+
# knos still tells people.
|
|
307
|
+
if hit.get("source") == "note" and not mem.remembered(
|
|
308
|
+
str(hit.get("about") or "")
|
|
309
|
+
):
|
|
310
|
+
continue
|
|
311
|
+
found.append(
|
|
312
|
+
Passage(
|
|
313
|
+
text=text,
|
|
314
|
+
source=str(hit.get("source") or hit.get("tier") or "note"),
|
|
315
|
+
where=str(hit.get("where") or hit.get("about") or "knos memory"),
|
|
316
|
+
path=str(hit.get("path") or ""),
|
|
317
|
+
score=_score(text, wanted)
|
|
318
|
+
+ (NOTE_LEAD if hit.get("source") == "note" else 0.0),
|
|
319
|
+
)
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
# Code structure is the slow source: the reader is a separate program and
|
|
323
|
+
# starting it costs seconds, against milliseconds for everything else.
|
|
324
|
+
# A question about a symbol or a call always pays that, because nothing
|
|
325
|
+
# in the sessions can answer it. A prose question pays it only when what
|
|
326
|
+
# was found is thin.
|
|
327
|
+
# Counted over what this caller may actually see. A teammate shared one
|
|
328
|
+
# folder has most of the repo filtered away, and judging "enough" on the
|
|
329
|
+
# part they cannot see left them with a grant that answered nothing.
|
|
330
|
+
seen_so_far = private.visible(repo, [p.__dict__ for p in found], identity, allowed)
|
|
331
|
+
thin = sum(1 for p in seen_so_far if p["score"] > 0) < ENOUGH
|
|
332
|
+
structural = looks_structural(question)
|
|
333
|
+
if (structural or thin) and code.installed():
|
|
334
|
+
for word in wanted[:3]:
|
|
335
|
+
try:
|
|
336
|
+
symbols, _ = code.search(repo, word, limit=10)
|
|
337
|
+
except Exception:
|
|
338
|
+
break # structure is a bonus source, never the reason to fail
|
|
339
|
+
for s in symbols:
|
|
340
|
+
found.append(
|
|
341
|
+
Passage(
|
|
342
|
+
text=f"{s.kind} {s.short}",
|
|
343
|
+
source="code",
|
|
344
|
+
where=s.where,
|
|
345
|
+
path=s.path,
|
|
346
|
+
# A question about the shape of the code wants the
|
|
347
|
+
# code first. Commit prose that merely mentions the
|
|
348
|
+
# word is background, however well it scores.
|
|
349
|
+
score=_score(f"{s.name} {s.kind}", wanted) + (STRUCTURAL_LEAD if structural else 0.0),
|
|
350
|
+
)
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
found = private.visible(repo, [p.__dict__ for p in found], identity, allowed)
|
|
354
|
+
passages = [Passage(**p) for p in found]
|
|
355
|
+
|
|
356
|
+
seen: set[str] = set()
|
|
357
|
+
ranked: list[Passage] = []
|
|
358
|
+
for p in sorted(passages, key=lambda p: -p.score):
|
|
359
|
+
if p.score <= 0:
|
|
360
|
+
continue
|
|
361
|
+
key = p.text[:120]
|
|
362
|
+
if key in seen:
|
|
363
|
+
continue
|
|
364
|
+
seen.add(key)
|
|
365
|
+
ranked.append(p)
|
|
366
|
+
if len(ranked) >= limit:
|
|
367
|
+
break
|
|
368
|
+
return ranked
|