knos 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
knos/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """knos — One memory for every coding agent on your machine, and it knows which of them is in your code right now."""
2
+
3
+ __version__ = "0.1.0"
knos/answer.py ADDED
@@ -0,0 +1,368 @@
1
+ """Reading a repo, and answering from what was read.
2
+
3
+ knos has no model and no API key. It does not write prose. An answer is the
4
+ passages it found and, beside each one, exactly where it came from: a file
5
+ and line, a session and date, or a commit. A developer will not trust it
6
+ twice without that.
7
+
8
+ Everything here runs because a person ran a command.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass
15
+ from pathlib import Path
16
+ from typing import Any
17
+
18
+ from . import code, errors, git, private
19
+ from .memory import INTERNAL, PERSON, Fact, Memory
20
+
21
+ STOP = {
22
+ "a", "about", "an", "and", "are", "as", "at", "be", "but", "by", "did",
23
+ "do", "does", "for", "from", "had", "has", "have", "how", "in", "is",
24
+ "it", "last", "of", "on", "or", "our", "that", "the", "their", "then",
25
+ "there", "this", "to", "was", "we", "were", "what", "when", "where",
26
+ "which", "who", "why", "with", "you", "your",
27
+ }
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class Passage:
32
+ """One thing knos found, and where it came from."""
33
+
34
+ text: str
35
+ source: str # "session" | "git" | "code" | "note"
36
+ where: str
37
+ score: float = 0.0
38
+ path: str = ""
39
+
40
+ @property
41
+ def line(self) -> str:
42
+ return f"{self.text}\n {self.where}"
43
+
44
+
45
+ # knos is not an archive. It keeps enough of a passage to recognise and
46
+ # search, and names where the whole thing still lives, so a person can go
47
+ # and read it in full. Storing less of each passage is always better than
48
+ # storing fewer of them: a repo whose oldest half is missing has holes a
49
+ # person cannot see, while shorter passages still point at everything.
50
+ KEEP_CHARS = 280
51
+
52
+ # How many good passages count as an answer. Below this, a prose question
53
+ # also asks the code reader, and pays the seconds that costs.
54
+ ENOUGH = 4
55
+
56
+ # What a deliberately written note is worth over derived chatter. Somebody
57
+ # chose to write it down; a session that happens to say the word twenty times
58
+ # did not. Without this the standing rule loses to the argument about it.
59
+ NOTE_LEAD = 1.0
60
+
61
+ # What a code result is worth over prose when the question was structural.
62
+ # Enough to outrank any single passage, so the file and line come first.
63
+ STRUCTURAL_LEAD = 2.0
64
+
65
+ # Ways of asking about the shape of the code rather than the story behind it.
66
+ # These always reach the code reader, however much prose was found, because
67
+ # no amount of session chatter answers "who calls this".
68
+ STRUCTURAL_PHRASES = (
69
+ "who calls",
70
+ "what calls",
71
+ "who uses",
72
+ "what uses",
73
+ "where is",
74
+ "where are",
75
+ "defined",
76
+ "definition",
77
+ "implemented",
78
+ "implementation",
79
+ "function",
80
+ "class",
81
+ "method",
82
+ "call graph",
83
+ "depends on",
84
+ "imports",
85
+ "signature",
86
+ )
87
+
88
+ # A name a programmer would type: CamelCase, snake_case, a call, or a file.
89
+ SYMBOL = re.compile(
90
+ r"[a-z]+_[a-z_]+|[a-z]+[A-Z]\w*|[A-Z][a-z]+[A-Z]\w*|\w+\(\)|[\w/\\.-]+\.[a-z]{1,4}\b"
91
+ )
92
+
93
+
94
+ def looks_structural(question: str) -> bool:
95
+ """Whether this asks about the shape of the code.
96
+
97
+ Asked of the question rather than of the results, because a question
98
+ about a symbol deserves the code reader even when the sessions happen to
99
+ be full of chatter that mentions it.
100
+ """
101
+ low = question.lower()
102
+ if any(phrase in low for phrase in STRUCTURAL_PHRASES):
103
+ return True
104
+ return bool(SYMBOL.search(question))
105
+
106
+
107
+ def topic_of(fact: str) -> str:
108
+ """What to file a fact under when nobody said.
109
+
110
+ The words that carry it, so "always use pnpm, never npm" files under
111
+ "pnpm npm" and is found by asking about either. Asking a person to
112
+ invent a filing name before they can write one sentence down is the
113
+ friction this exists to avoid.
114
+ """
115
+ # A standing instruction is mostly instruction. What it is *about* is
116
+ # what is left once the telling-off is removed.
117
+ telling = {
118
+ "always", "never", "must", "should", "dont", "please", "make",
119
+ "sure", "keep", "stop", "avoid", "prefer", "only", "ever", "when",
120
+ "use", "using", "used", "into", "onto", "over", "under", "before",
121
+ "after", "instead", "rather", "than", "them", "they",
122
+ }
123
+ # Not a length filter: npm, git, ssh and api are all three letters and
124
+ # all exactly what a rule is about.
125
+ words = [w for w in terms(fact) if w not in telling][:3]
126
+ return " ".join(words) or fact[:40].strip()
127
+
128
+
129
+ def _trim(text: str) -> str:
130
+ text = " ".join(text.split())
131
+ if len(text) <= KEEP_CHARS:
132
+ return text
133
+ return text[:KEEP_CHARS].rsplit(" ", 1)[0] + "..."
134
+
135
+
136
+ # ---- reading a repo ---------------------------------------------------
137
+
138
+
139
+ def point(
140
+ repo: Path,
141
+ mem: Memory,
142
+ index_code: bool = True,
143
+ on_progress: Any = None,
144
+ ) -> dict[str, int]:
145
+ """Read this repo: its sessions, its commits, its structure.
146
+
147
+ Called by `knos point`. Facts are stated as they were found; nothing is
148
+ summarised, scored or inferred.
149
+ """
150
+ from . import sessions
151
+
152
+ counts: dict = {
153
+ "sessions": 0,
154
+ "commits": 0,
155
+ "code": 0,
156
+ "private": 0,
157
+ "full": 0,
158
+ "skipped": [],
159
+ }
160
+ say = on_progress or (lambda *_: None)
161
+ repo = Path(repo).resolve()
162
+
163
+ mem.set_reference(INTERNAL + "repo", {"path": str(repo), "name": repo.name})
164
+
165
+ # Newest first, so that if the store fills up what knos kept is the part
166
+ # anyone is likely to ask about.
167
+ say("looking for past agent sessions")
168
+ for turn in reversed(sessions.read_all(repo)):
169
+ if not mem.record(
170
+ Fact(
171
+ text=_trim(turn.text),
172
+ source="session",
173
+ where=turn.where,
174
+ when=turn.when,
175
+ about=turn.client,
176
+ )
177
+ ):
178
+ counts["full"] = 1
179
+ break
180
+ counts["sessions"] += 1
181
+ if counts["sessions"] % 100 == 0:
182
+ say(f"{counts['sessions']} things said in past sessions")
183
+
184
+ if not counts["full"]:
185
+ # Commits arrive newest first, so the first time a file or a person
186
+ # is seen is their latest. A canonical record is written once, then
187
+ # left alone: rewriting it for every older commit produced the same
188
+ # answer after tens of thousands of pointless writes, which was
189
+ # most of what `knos point` spent its time doing.
190
+ named: set[str] = set()
191
+ say("reading the commits")
192
+ for commit in git.read_commits(repo):
193
+ visible = [f for f in commit.files if not private.is_private(repo, f)]
194
+ counts["private"] += len(commit.files) - len(visible)
195
+ if not mem.record(
196
+ Fact(
197
+ text=_trim(commit.text),
198
+ source="git",
199
+ where=commit.where,
200
+ when=commit.when,
201
+ about=commit.author,
202
+ path=visible[0] if visible else "",
203
+ )
204
+ ):
205
+ counts["full"] = 1
206
+ break
207
+ # Only people get a canonical record here. A file used to get one
208
+ # per commit that touched it, which on a real repo was hundreds of
209
+ # writes at twenty-five milliseconds each, for something the
210
+ # journal already says: the commit that changed a file names the
211
+ # file. That was most of what reading a repo cost.
212
+ if commit.author not in named:
213
+ named.add(commit.author)
214
+ mem.note_thing(
215
+ PERSON,
216
+ commit.author,
217
+ {"last_commit": commit.short, "last_seen": commit.when[:10]},
218
+ )
219
+ counts["commits"] += 1
220
+ if counts["commits"] % 100 == 0:
221
+ say(f"{counts['commits']} commits")
222
+
223
+ if index_code and code.installed():
224
+ result: dict = {}
225
+ # The longest stretch, and the one with nothing to count as it
226
+ # goes, so say plainly that a wait here is expected.
227
+ say("reading the code, the slow part")
228
+ try:
229
+ result = code.index(repo)
230
+ except Exception:
231
+ # Structure is one source of three. Losing it is worth a line,
232
+ # not the whole command.
233
+ counts["skipped"].append(
234
+ errors.unreadable("code structure", "the reader would not start")
235
+ )
236
+ counts["code"] = int(result.get("nodes") or 0)
237
+ mem.set_reference(INTERNAL + "code_index", result)
238
+
239
+ # What goes into the store has to be plain data, so the skipped files are
240
+ # counted here and handed back to the caller in full.
241
+ mem.set_focus(
242
+ {
243
+ "repo": str(repo),
244
+ "read": {k: v for k, v in counts.items() if isinstance(v, int)},
245
+ "skipped": len(counts["skipped"]),
246
+ }
247
+ )
248
+ return counts
249
+
250
+
251
+ # ---- answering --------------------------------------------------------
252
+
253
+
254
+ def terms(question: str) -> list[str]:
255
+ words = re.findall(r"[A-Za-z_][A-Za-z0-9_.\-]{1,}", question.lower())
256
+ return [w for w in words if w not in STOP and len(w) > 2]
257
+
258
+
259
+ def _score(text: str, wanted: list[str]) -> float:
260
+ low = text.lower()
261
+ hits = sum(1 for w in wanted if w in low)
262
+ if not hits:
263
+ return 0.0
264
+ # Prefer passages that cover more of the question over ones that repeat
265
+ # a single word, and prefer shorter passages at equal coverage.
266
+ return hits / len(wanted) + min(len(low), 2000) / 200000.0
267
+
268
+
269
+ def ask(
270
+ repo: Path,
271
+ mem: Memory,
272
+ question: str,
273
+ identity: str = private.OWNER,
274
+ limit: int = 8,
275
+ allowed: list[str] | None = None,
276
+ ) -> list[Passage]:
277
+ """Ranked passages with their sources. No prose, no synthesis.
278
+
279
+ For an agent, private paths are dropped before ranking and are never
280
+ counted, so the reply carries no sign that anything was withheld.
281
+ """
282
+ wanted = terms(question)
283
+ if not wanted:
284
+ return []
285
+
286
+ found: list[Passage] = []
287
+
288
+ # The store's full-text search wants terms, not a sentence: asking it for
289
+ # the whole question makes every word mandatory and finds nothing. Each
290
+ # term is asked for separately and the results are ranked together, so a
291
+ # passage that covers more of the question wins.
292
+ for term in wanted[:6]:
293
+ for hit in mem.search(term, limit=60):
294
+ text = str(hit.get("text") or "").strip()
295
+ if not text:
296
+ continue
297
+ # A canonical record says what knos knows *about* a thing, not
298
+ # where it heard it, so it has no source to name and no business
299
+ # being quoted as an answer. The journal carries the same fact
300
+ # with the commit it came from. `about` is where these belong.
301
+ if hit.get("tier") == "entity":
302
+ continue
303
+ # A note that has been forgotten stops being an answer. The
304
+ # journal keeps the line, because the journal is a record of what
305
+ # was learned and when, but a forgotten thing is not something
306
+ # knos still tells people.
307
+ if hit.get("source") == "note" and not mem.remembered(
308
+ str(hit.get("about") or "")
309
+ ):
310
+ continue
311
+ found.append(
312
+ Passage(
313
+ text=text,
314
+ source=str(hit.get("source") or hit.get("tier") or "note"),
315
+ where=str(hit.get("where") or hit.get("about") or "knos memory"),
316
+ path=str(hit.get("path") or ""),
317
+ score=_score(text, wanted)
318
+ + (NOTE_LEAD if hit.get("source") == "note" else 0.0),
319
+ )
320
+ )
321
+
322
+ # Code structure is the slow source: the reader is a separate program and
323
+ # starting it costs seconds, against milliseconds for everything else.
324
+ # A question about a symbol or a call always pays that, because nothing
325
+ # in the sessions can answer it. A prose question pays it only when what
326
+ # was found is thin.
327
+ # Counted over what this caller may actually see. A teammate shared one
328
+ # folder has most of the repo filtered away, and judging "enough" on the
329
+ # part they cannot see left them with a grant that answered nothing.
330
+ seen_so_far = private.visible(repo, [p.__dict__ for p in found], identity, allowed)
331
+ thin = sum(1 for p in seen_so_far if p["score"] > 0) < ENOUGH
332
+ structural = looks_structural(question)
333
+ if (structural or thin) and code.installed():
334
+ for word in wanted[:3]:
335
+ try:
336
+ symbols, _ = code.search(repo, word, limit=10)
337
+ except Exception:
338
+ break # structure is a bonus source, never the reason to fail
339
+ for s in symbols:
340
+ found.append(
341
+ Passage(
342
+ text=f"{s.kind} {s.short}",
343
+ source="code",
344
+ where=s.where,
345
+ path=s.path,
346
+ # A question about the shape of the code wants the
347
+ # code first. Commit prose that merely mentions the
348
+ # word is background, however well it scores.
349
+ score=_score(f"{s.name} {s.kind}", wanted) + (STRUCTURAL_LEAD if structural else 0.0),
350
+ )
351
+ )
352
+
353
+ found = private.visible(repo, [p.__dict__ for p in found], identity, allowed)
354
+ passages = [Passage(**p) for p in found]
355
+
356
+ seen: set[str] = set()
357
+ ranked: list[Passage] = []
358
+ for p in sorted(passages, key=lambda p: -p.score):
359
+ if p.score <= 0:
360
+ continue
361
+ key = p.text[:120]
362
+ if key in seen:
363
+ continue
364
+ seen.add(key)
365
+ ranked.append(p)
366
+ if len(ranked) >= limit:
367
+ break
368
+ return ranked