pyacri 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
acri/__init__.py ADDED
@@ -0,0 +1,73 @@
1
+ """acri — a client-side capability resolver. `import acri` is the whole product.
2
+
3
+ `run()` is the one function that touches every module: corpus's index,
4
+ compass's ranking, port's provider call, and ledger's trace. Integration
5
+ tests call this, not a hand-wired re-implementation of the pipeline.
6
+
7
+ Named `__init__.py`, not `kernel.py`: docs/architecture.md #5 retired
8
+ kernel/runtime/OS vocabulary for this project on purpose. The entry point
9
+ doesn't get to bring it back through the filename.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import time
14
+ from typing import Any
15
+
16
+ from .adapters import from_callables, from_mcp_tools
17
+ from .compass import Resolved
18
+ from .compass import resolve as _compass_resolve
19
+ from .corpus import Corpus, Tool, index
20
+ from .escape_hatch import FIND_MORE_TOOLS, find_more_tools
21
+ from .ledger import Entry, Ledger
22
+ from .port import GenerationResult, cached_call, gemini, openai_compatible
23
+ from .router import route
24
+
25
+ __all__ = [
26
+ "Tool", "Corpus", "index", "from_callables", "from_mcp_tools",
27
+ "Resolved", "resolve",
28
+ "FIND_MORE_TOOLS", "find_more_tools",
29
+ "GenerationResult", "gemini", "openai_compatible",
30
+ "Entry", "Ledger",
31
+ "run",
32
+ ]
33
+
34
+ _PROVIDERS = {"openai": openai_compatible, "gemini": gemini}
35
+
36
+
37
+ def resolve(query: str, corpus: Corpus, k: int = 5) -> list[Resolved]:
38
+ """Rank `corpus` against `query`, return the top k. See compass.resolve for the algorithm."""
39
+ return _compass_resolve(query, corpus, k)
40
+
41
+
42
+ def run(
43
+ query: str,
44
+ corpus: Corpus,
45
+ client: Any,
46
+ provider: str = "openai",
47
+ *,
48
+ k: int = 5,
49
+ model: str | None = None,
50
+ cheap_model: str | None = None,
51
+ ledger: Ledger | None = None,
52
+ cache: dict[Any, GenerationResult] | None = None,
53
+ ) -> GenerationResult:
54
+ """Resolve tools for `query`, call `provider` with them, log the trace if a ledger is given.
55
+
56
+ `cache` (a dict) skips a repeated (provider, model, query, offered tools) call -- see
57
+ port.cached_call, decisions.md #8c. `cheap_model` routes this one call to a cheaper
58
+ tier -- see router.route, decisions.md #1.
59
+ """
60
+ call = _PROVIDERS.get(provider)
61
+ if call is None:
62
+ raise ValueError(f"unknown provider: {provider!r} (expected one of {sorted(_PROVIDERS)})")
63
+ model = route(model, cheap_model)
64
+ start = time.time()
65
+ resolved = _compass_resolve(query, corpus, k)
66
+ offered = [*resolved, Resolved(tool=FIND_MORE_TOOLS, score=0.0)]
67
+ key = (provider, model, query, tuple(r.tool.name for r in resolved))
68
+ kwargs = {"model": model} if model else {}
69
+ result = cached_call(call, cache, key, client, query, offered, **kwargs)
70
+ if ledger is not None:
71
+ selected = [c["name"] for c in result.tool_calls]
72
+ ledger.record(query, resolved, selected, (time.time() - start) * 1000, corpus_size=len(corpus))
73
+ return result
acri/_synonyms.py ADDED
@@ -0,0 +1,31 @@
1
+ """_synonyms — query-side alias expansion for compass. Not corpus-facing.
2
+
3
+ Words a user types that never literally appear in any tool description --
4
+ BM25 has no stemming or semantic understanding, so "rain" and "weather" are
5
+ unrelated tokens to it otherwise. Applied to the query only, never the
6
+ corpus: expanding doc text here would let a tool's description quietly
7
+ start matching queries for a synonym it never claimed, and would shift
8
+ df/idf for every other tool too. Every entry traces to a real recall@5
9
+ miss, not a guess -- see assay/diagnose.py.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ ALIASES: dict[str, tuple[str, ...]] = {
14
+ "pr": ("pull", "request"),
15
+ "meeting": ("event",),
16
+ "rain": ("weather",),
17
+ "raining": ("weather",),
18
+ "storm": ("weather",),
19
+ "warning": ("alerts",),
20
+ "warnings": ("alerts",),
21
+ "text": ("sms", "message"),
22
+ "sharpen": ("upscale", "resolution"),
23
+ "money": ("refund", "charge"),
24
+ }
25
+
26
+
27
+ def expand(tokens: list[str]) -> list[str]:
28
+ expanded = list(tokens)
29
+ for tok in tokens:
30
+ expanded.extend(ALIASES.get(tok, ()))
31
+ return expanded
acri/_template.py ADDED
@@ -0,0 +1,27 @@
1
+ """_template — the acri init scaffold. Split out of cli.py (word-cap): this
2
+ is static content, not CLI parsing/dispatch logic.
3
+ """
4
+ from __future__ import annotations
5
+
6
+ TEMPLATE = """\
7
+ version: 1
8
+
9
+ models:
10
+ default: gemini-2.5-flash
11
+ # cheap: a stateless, prefix-free tier only -- classification, extraction,
12
+ # summarizing a tool result. See docs/architecture.md #4.4. Optional.
13
+ # cheap: gemini-2.5-flash-lite
14
+
15
+ mcp:
16
+ # - name: github
17
+ # command: ["npx", "-y", "@modelcontextprotocol/server-github"]
18
+ # - name: postgres
19
+ # url: http://localhost:3001
20
+
21
+ resolve:
22
+ k: 5
23
+
24
+ limits:
25
+ timeout_ms: 5000
26
+ max_cost_per_task_usd: 0.05
27
+ """
acri/_text.py ADDED
@@ -0,0 +1,34 @@
1
+ """_text — tokenization shared by corpus (indexing) and compass (scoring).
2
+
3
+ Internal to the package: both sides must tokenize identically or BM25 scores
4
+ against a corpus the query was never consistently split against.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import re
9
+
10
+ _TOKEN_RE = re.compile(r"[a-z0-9]+")
11
+
12
+ # Standard English function words (the NLTK stoplist's non-contraction
13
+ # core), filtered before scoring. Started as a smaller ad-hoc set; expanded
14
+ # to this external, non-cherry-picked list after "into" — missing from the
15
+ # original set — tied github_merge_pull_request with translate/salesforce
16
+ # tools that happened to also contain "into". A hand-patched one-word-at-a-
17
+ # time list invites tuning to whichever queries you happen to be looking at;
18
+ # a standard external stoplist doesn't. See assay/diagnose.py.
19
+ _STOPWORDS = frozenset("""
20
+ a an the is are was were be been being am of in on at to for with and or
21
+ but this that these those what which who whom me my you your it its we
22
+ they do does did into from by about up down out off over under again
23
+ further then once here there when where why how all any both each few
24
+ more most other some such no nor not only own same so than too very s t
25
+ can will just now if because as until while against between during
26
+ before after above below
27
+ """.split())
28
+
29
+
30
+ def tokenize(text: str) -> list[str]:
31
+ # Strip apostrophes before splitting: "user's" -> "users" as one token,
32
+ # not "user" + a stray "s" that then matches every other possessive.
33
+ text = text.lower().replace("'", "").replace("’", "")
34
+ return [t for t in _TOKEN_RE.findall(text) if t not in _STOPWORDS]
acri/adapters.py ADDED
@@ -0,0 +1,50 @@
1
+ """adapters — turn external tool descriptions into `corpus.Tool`."""
2
+ from __future__ import annotations
3
+
4
+ import inspect
5
+ from typing import Any, Callable
6
+
7
+ from .corpus import Tool
8
+
9
+ _TYPE_MAP = {str: "string", int: "integer", float: "number", bool: "boolean"}
10
+
11
+
12
+ def from_mcp_tools(mcp_tools: list[dict[str, Any]]) -> list[Tool]:
13
+ """Adapt an MCP `tools/list` response into Tools. Feed the result to `index()`."""
14
+ return [
15
+ Tool(
16
+ name=t["name"],
17
+ description=t.get("description", ""),
18
+ parameters=t.get("inputSchema", {"type": "object", "properties": {}}),
19
+ )
20
+ for t in mcp_tools
21
+ ]
22
+
23
+
24
+ def _schema_from_callable(f: Callable[..., Any]) -> dict[str, Any]:
25
+ sig = inspect.signature(f)
26
+ props: dict[str, Any] = {}
27
+ required: list[str] = []
28
+ for name, param in sig.parameters.items():
29
+ if param.kind in (param.VAR_POSITIONAL, param.VAR_KEYWORD):
30
+ continue
31
+ props[name] = {"type": _TYPE_MAP.get(param.annotation, "string")}
32
+ if param.default is inspect.Parameter.empty:
33
+ required.append(name)
34
+ schema: dict[str, Any] = {"type": "object", "properties": props}
35
+ if required:
36
+ schema["required"] = required
37
+ return schema
38
+
39
+
40
+ def from_callables(funcs: list[Callable[..., Any]]) -> list[Tool]:
41
+ """Adapt plain Python functions into Tools, deriving each schema from its signature."""
42
+ return [
43
+ Tool(
44
+ name=f.__name__,
45
+ description=(f.__doc__ or "").strip(),
46
+ parameters=_schema_from_callable(f),
47
+ handler=f,
48
+ )
49
+ for f in funcs
50
+ ]
acri/cli.py ADDED
@@ -0,0 +1,97 @@
1
+ """cli — `acri init` writes a template acri.yaml; `acri check` validates one;
2
+ `acri up` runs the daemon; `acri studio` runs the read-only dashboard. Both
3
+ ship ahead of decisions.md's own gates, at the maintainer's request -- see
4
+ acri/server.py and acri/studio.py for specifics.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ from ._template import TEMPLATE
13
+
14
+
15
+ def _init(args: argparse.Namespace) -> int:
16
+ path = Path(args.path)
17
+ if path.exists():
18
+ print(f"{path} already exists -- not overwriting.", file=sys.stderr)
19
+ return 1
20
+ path.write_text(TEMPLATE, encoding="utf-8")
21
+ print(f"wrote {path}")
22
+ return 0
23
+
24
+
25
+ def _check(args: argparse.Namespace) -> int:
26
+ try:
27
+ from .config import from_yaml # lazy: needs pyacri[yaml]
28
+ from .credentials import missing_env_vars
29
+ except ImportError:
30
+ print("acri check needs PyYAML -- pip install pyacri[yaml]", file=sys.stderr)
31
+ return 1
32
+ path = Path(args.path)
33
+ if not path.exists():
34
+ print(f"{path} not found -- run `acri init` first.", file=sys.stderr)
35
+ return 1
36
+ missing = missing_env_vars(from_yaml(path))
37
+ if missing:
38
+ print("missing credentials:")
39
+ for var in missing:
40
+ print(f" {var}")
41
+ return 1
42
+ print(f"ok - {path} is valid, all credentials present")
43
+ return 0
44
+
45
+
46
+ def _up(args: argparse.Namespace) -> int:
47
+ try:
48
+ from .server import serve # lazy: needs pyacri[server]
49
+ except ImportError:
50
+ print("acri up needs mcp and PyYAML -- pip install pyacri[server]", file=sys.stderr)
51
+ return 1
52
+ serve(args.path, host=args.host, port=args.port, log_conversations=args.log_conversations)
53
+ return 0
54
+
55
+
56
+ def _studio(args: argparse.Namespace) -> int:
57
+ try:
58
+ from .studio import serve_studio # lazy: needs pyacri[yaml]
59
+ except ImportError:
60
+ print("acri studio needs PyYAML -- pip install pyacri[studio]", file=sys.stderr)
61
+ return 1
62
+ serve_studio(args.path, ledger_path=args.ledger, host=args.host, port=args.port)
63
+ return 0
64
+
65
+
66
+ def main(argv: list[str] | None = None) -> int:
67
+ parser = argparse.ArgumentParser(prog="acri")
68
+ sub = parser.add_subparsers(dest="command", required=True)
69
+
70
+ p_init = sub.add_parser("init", help="write a template acri.yaml")
71
+ p_init.add_argument("path", nargs="?", default="acri.yaml")
72
+ p_init.set_defaults(func=_init)
73
+
74
+ p_check = sub.add_parser("check", help="validate acri.yaml and its credentials")
75
+ p_check.add_argument("path", nargs="?", default="acri.yaml")
76
+ p_check.set_defaults(func=_check)
77
+
78
+ p_up = sub.add_parser("up", help="run the daemon: an OpenAI-compatible endpoint over acri.yaml's tools")
79
+ p_up.add_argument("path", nargs="?", default="acri.yaml")
80
+ p_up.add_argument("--host", default="127.0.0.1")
81
+ p_up.add_argument("--port", type=int, default=8080)
82
+ p_up.add_argument("--log-conversations", action="store_true")
83
+ p_up.set_defaults(func=_up)
84
+
85
+ p_studio = sub.add_parser("studio", help="run the read-only dashboard over acri.yaml and the ledger")
86
+ p_studio.add_argument("path", nargs="?", default="acri.yaml")
87
+ p_studio.add_argument("--ledger", default=".acri/ledger.jsonl")
88
+ p_studio.add_argument("--host", default="127.0.0.1")
89
+ p_studio.add_argument("--port", type=int, default=8099)
90
+ p_studio.set_defaults(func=_studio)
91
+
92
+ args = parser.parse_args(argv)
93
+ return args.func(args)
94
+
95
+
96
+ if __name__ == "__main__":
97
+ sys.exit(main())
acri/compass.py ADDED
@@ -0,0 +1,70 @@
1
+ """compass — the resolver. Given a query, returns the k tools that matter.
2
+
3
+ This is the product. It performs no language understanding — BM25 scores
4
+ query terms against tool text, weighting terms that are rare in the corpus.
5
+ That's deliberate: understanding is the model's job and costs a model call.
6
+ compass only does recall (narrow N tools to k candidates); the model still
7
+ does precision (pick one, write its arguments). See docs/decisions.md.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import math
12
+ from dataclasses import dataclass
13
+
14
+ from ._synonyms import expand as _expand
15
+ from ._text import tokenize as _tokenize
16
+ from .corpus import Corpus, Tool
17
+
18
+ _K1 = 1.5
19
+ _B = 0.75
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class Resolved:
24
+ """One ranked tool and acri's confidence in it.
25
+
26
+ `score` is relative to the best match for this query (1.0 = best), not a
27
+ calibrated probability — a 0.9 does not mean "90% sure".
28
+ """
29
+
30
+ tool: Tool
31
+ score: float
32
+
33
+
34
+ def _idf(df: int, n_docs: int) -> float:
35
+ return math.log(1 + (n_docs - df + 0.5) / (df + 0.5))
36
+
37
+
38
+ def bm25(query_tokens: list[str], corpus: Corpus, doc_idx: int) -> float:
39
+ doc_len = len(corpus.doc_tokens[doc_idx])
40
+ freqs = corpus.doc_freqs[doc_idx]
41
+ n_docs = len(corpus.tools)
42
+ score = 0.0
43
+ for tok in query_tokens:
44
+ f = freqs.get(tok)
45
+ if not f:
46
+ continue
47
+ idf = _idf(corpus.df.get(tok, 0), n_docs)
48
+ denom = f + _K1 * (1 - _B + _B * doc_len / corpus.avgdl)
49
+ score += idf * (f * (_K1 + 1)) / denom
50
+ return score
51
+
52
+
53
+ def resolve(query: str, corpus: Corpus, k: int = 5) -> list[Resolved]:
54
+ """Rank every tool in `corpus` against `query`, return the top k.
55
+
56
+ Tools that score zero (no shared term with the query) are dropped rather
57
+ than padded in — an empty result means "nothing in this corpus matches",
58
+ which the caller should handle via `port`'s no-tools path, not treat as
59
+ an error.
60
+ """
61
+ if len(corpus) == 0:
62
+ return []
63
+ query_tokens = _expand(_tokenize(query))
64
+ raw = [bm25(query_tokens, corpus, i) for i in range(len(corpus))]
65
+ top = max(raw, default=0.0)
66
+ if top <= 0.0:
67
+ return []
68
+ scored = [Resolved(tool=corpus.tools[i], score=raw[i] / top) for i in range(len(corpus)) if raw[i] > 0.0]
69
+ scored.sort(key=lambda r: r.score, reverse=True)
70
+ return scored[:k]
acri/config.py ADDED
@@ -0,0 +1,87 @@
1
+ """config — parses acri.yaml. docs/decisions.md: "declarative configuration".
2
+
3
+ Declares capabilities and limits, never control flow -- no steps/on_error/
4
+ conditionals, on purpose (see decisions.md). Requires PyYAML: `pip install
5
+ pyacri[yaml]`. Not imported from acri/__init__.py, so `import acri` alone never
6
+ needs it -- the core library stays dependency-free.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ import yaml
15
+
16
+ _SUPPORTED_VERSION = 1
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class ModelsConfig:
21
+ default: str | None = None
22
+ cheap: str | None = None
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class SandboxConfig: # v1.1 resource/network/volume limits -- see acri.sandbox.sandboxed()
27
+ image: str
28
+ memory: str = "256m"
29
+ cpus: float = 0.5
30
+ network: bool = True
31
+ volumes: dict[str, str] | None = None # host path -> container path
32
+
33
+
34
+ @dataclass(frozen=True)
35
+ class McpEntry:
36
+ name: str
37
+ command: list[str] | None = None
38
+ url: str | None = None
39
+ sandbox: SandboxConfig | None = None # command: entries only -- see _mcp_entry
40
+
41
+
42
+ @dataclass(frozen=True)
43
+ class Limits:
44
+ timeout_ms: int | None = None
45
+ max_cost_per_task_usd: float | None = None
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class Config:
50
+ """Parsed acri.yaml. `k` and `limits` mirror acri.run()'s own parameters --
51
+ limits aren't enforced yet (that's v1.0's held-back remainder)."""
52
+
53
+ version: int
54
+ models: ModelsConfig = field(default_factory=ModelsConfig)
55
+ mcp: list[McpEntry] = field(default_factory=list)
56
+ k: int = 5
57
+ limits: Limits = field(default_factory=Limits)
58
+
59
+
60
+ def _mcp_entry(raw: dict[str, Any]) -> McpEntry:
61
+ command, url = raw.get("command"), raw.get("url")
62
+ if bool(command) == bool(url):
63
+ raise ValueError(f"mcp entry {raw.get('name')!r} needs exactly one of command or url")
64
+ sandbox = raw.get("sandbox")
65
+ if sandbox and not command:
66
+ raise ValueError(f"mcp entry {raw.get('name')!r}: sandbox needs command, not url")
67
+ return McpEntry(name=raw["name"], command=command, url=url,
68
+ sandbox=SandboxConfig(**sandbox) if sandbox else None)
69
+
70
+
71
+ def from_yaml(path: str | Path) -> Config:
72
+ """Load and validate an acri.yaml. Raises ValueError on an unsupported
73
+ version or a malformed mcp entry; yaml.YAMLError on invalid YAML."""
74
+ data = yaml.safe_load(Path(path).read_text(encoding="utf-8")) or {}
75
+ version = data.get("version")
76
+ if version != _SUPPORTED_VERSION:
77
+ raise ValueError(f"unsupported acri.yaml version: {version!r} (expected {_SUPPORTED_VERSION})")
78
+ models = data.get("models") or {}
79
+ resolve = data.get("resolve") or {}
80
+ limits = data.get("limits") or {}
81
+ return Config(
82
+ version=version,
83
+ models=ModelsConfig(default=models.get("default"), cheap=models.get("cheap")),
84
+ mcp=[_mcp_entry(m) for m in data.get("mcp") or []],
85
+ k=resolve.get("k", 5),
86
+ limits=Limits(timeout_ms=limits.get("timeout_ms"), max_cost_per_task_usd=limits.get("max_cost_per_task_usd")),
87
+ )
acri/corpus.py ADDED
@@ -0,0 +1,57 @@
1
+ """corpus — the capability index. Ingests tools into one searchable body, once."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass, field
5
+ from typing import Any, Callable
6
+
7
+ from ._text import tokenize as _tokenize
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Tool:
12
+ """One capability: a name, a description, a JSON Schema, and how to reach it.
13
+
14
+ `handler` is optional — resolution never calls it. It's a place for the
15
+ caller (or a future dispatcher) to find the actual function.
16
+ """
17
+
18
+ name: str
19
+ description: str
20
+ parameters: dict[str, Any] = field(default_factory=lambda: {"type": "object", "properties": {}})
21
+ handler: Callable[..., Any] | None = None
22
+
23
+
24
+ @dataclass
25
+ class Corpus:
26
+ """A built capability index. Construct with `index()`, not directly.
27
+
28
+ The four fields below are the prebuilt BM25 index that `compass.resolve`
29
+ reads — treat this as an opaque handle unless you're changing the ranking.
30
+ """
31
+
32
+ tools: list[Tool]
33
+ doc_tokens: list[list[str]]
34
+ doc_freqs: list[dict[str, int]]
35
+ df: dict[str, int]
36
+ avgdl: float
37
+
38
+ def __len__(self) -> int:
39
+ return len(self.tools)
40
+
41
+
42
+ def index(tools: list[Tool]) -> Corpus:
43
+ """Build a searchable Corpus once. Reuse it across turns — never rebuild per query."""
44
+ if not tools:
45
+ raise ValueError("index() needs at least one tool")
46
+ doc_tokens = [_tokenize(f"{t.name} {t.description}") for t in tools]
47
+ doc_freqs: list[dict[str, int]] = []
48
+ df: dict[str, int] = {}
49
+ for tokens in doc_tokens:
50
+ freqs: dict[str, int] = {}
51
+ for tok in tokens:
52
+ freqs[tok] = freqs.get(tok, 0) + 1
53
+ doc_freqs.append(freqs)
54
+ for tok in freqs:
55
+ df[tok] = df.get(tok, 0) + 1
56
+ avgdl = sum(len(t) for t in doc_tokens) / len(doc_tokens)
57
+ return Corpus(tools=list(tools), doc_tokens=doc_tokens, doc_freqs=doc_freqs, df=df, avgdl=avgdl)
acri/credentials.py ADDED
@@ -0,0 +1,25 @@
1
+ """credentials — which env vars a Config's models need, and which are missing.
2
+
3
+ Split out of config.py: parsing acri.yaml and checking credentials are
4
+ different concerns that happen to both feed `acri check`.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import os
9
+
10
+ from .config import Config
11
+
12
+ _ENV_VARS = {"gemini": "GEMINI_API_KEY", "openai": "OPENAI_API_KEY"}
13
+
14
+
15
+ def provider_for(model: str) -> str:
16
+ return "gemini" if "gemini" in model.lower() else "openai"
17
+
18
+
19
+ def missing_env_vars(config: Config) -> list[str]:
20
+ """Standard env vars (GEMINI_API_KEY, OPENAI_API_KEY) any configured model needs,
21
+ that aren't set. Inferred from the model name, not a `provider:` field -- the
22
+ documented acri.yaml example doesn't have one."""
23
+ models = [m for m in (config.models.default, config.models.cheap) if m]
24
+ needed = {_ENV_VARS[provider_for(m)] for m in models}
25
+ return sorted(v for v in needed if not os.environ.get(v))
acri/daemon.py ADDED
@@ -0,0 +1,75 @@
1
+ """daemon — the request handler `acri up` will eventually serve over HTTP.
2
+
3
+ No socket here, on purpose: this is the thin OpenAI-shaped layer over
4
+ acri.run(), kept separately testable so "the daemon is a thin wrapper over
5
+ the library, never a superset" (docs/decisions.md) is a test, not just a
6
+ sentence. Wiring this into a real listening process is v1.0's gated part.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from . import run
14
+ from .corpus import Corpus
15
+ from .ledger import Ledger
16
+ from .port import GenerationResult
17
+
18
+ DEFAULT_LEDGER_PATH = Path(".acri/ledger.jsonl")
19
+
20
+
21
+ def default_ledger(path: Path | str = DEFAULT_LEDGER_PATH) -> Ledger:
22
+ """A Ledger backed by `.acri/ledger.jsonl` (docs/decisions.md), creating the
23
+ directory if it doesn't exist yet."""
24
+ path = Path(path)
25
+ path.parent.mkdir(parents=True, exist_ok=True)
26
+ return Ledger(path)
27
+
28
+
29
+ class RedactingLedger:
30
+ """Wraps a real Ledger, dropping the query text before it's recorded.
31
+
32
+ docs/decisions.md: "ledger records decisions, scores, and token counts.
33
+ Conversation content is opt-in." -- `acri up`'s default. Duck-typed to
34
+ Ledger.record's signature, nothing else; `run()` never checks the type.
35
+ """
36
+
37
+ def __init__(self, ledger: Ledger) -> None:
38
+ self._ledger = ledger
39
+
40
+ def record(self, query: str, offered: Any, selected: list[str], latency_ms: float, cost_usd: float | None = None) -> Any:
41
+ return self._ledger.record("<redacted>", offered, selected, latency_ms, cost_usd)
42
+
43
+
44
+ def handle_chat_completion(
45
+ request: dict[str, Any],
46
+ corpus: Corpus,
47
+ client: Any,
48
+ provider: str = "openai",
49
+ *,
50
+ k: int = 5,
51
+ cheap_model: str | None = None,
52
+ ledger: Ledger | None = None,
53
+ cache: dict[Any, GenerationResult] | None = None,
54
+ ) -> dict[str, Any]:
55
+ """Handle one OpenAI-shaped `/v1/chat/completions` request via acri.run() --
56
+ not a reimplementation of resolve+call.
57
+
58
+ Single-turn only: the last message's content is the query. `acri/server.py`
59
+ wraps this in SSE at the wire level; multi-turn history and true upstream
60
+ streaming (vs. one blocking call chunked out) are still open.
61
+ """
62
+ messages = request.get("messages") or []
63
+ query = messages[-1]["content"] if messages else ""
64
+ result = run(
65
+ query, corpus, client, provider,
66
+ k=k, model=request.get("model"), cheap_model=cheap_model,
67
+ ledger=ledger, cache=cache,
68
+ )
69
+ message: dict[str, Any] = {"role": "assistant", "content": result.text}
70
+ if result.tool_calls:
71
+ message["tool_calls"] = [
72
+ {"id": f"call_{i}", "type": "function", "function": tc}
73
+ for i, tc in enumerate(result.tool_calls)
74
+ ]
75
+ return {"choices": [{"message": message}]}
acri/escape_hatch.py ADDED
@@ -0,0 +1,38 @@
1
+ """escape_hatch — find_more_tools, the recovery path architecture.md #4.1 names.
2
+
3
+ `compass.resolve()` picks k tools once per task and that resolution is never
4
+ rewritten (see compass.py, decisions.md #4.1: rewriting a sent prefix costs
5
+ more than doing nothing). But a bad initial resolution still has to be
6
+ recoverable, so `compass` "always includes a find_more_tools capability" --
7
+ this is that capability, kept out of compass.py so resolve()'s own contract
8
+ (pure ranking, nothing appended) stays untouched and every recall@k number
9
+ in assay/ stays valid.
10
+
11
+ `acri.run()` appends FIND_MORE_TOOLS to what's actually sent to the provider,
12
+ not to the ledger's `offered` -- the ledger records what compass resolved,
13
+ not the constant scaffolding wrapped around it. Calling it re-searches the
14
+ full corpus; acri never auto-executes it, same as any other Tool.handler.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ from .compass import resolve
19
+ from .corpus import Corpus, Tool
20
+
21
+ FIND_MORE_TOOLS = Tool(
22
+ name="find_more_tools",
23
+ description="Search for tools beyond the ones already offered, when none of them fit this task.",
24
+ parameters={
25
+ "type": "object",
26
+ "properties": {"query": {"type": "string", "description": "what capability you need"}},
27
+ "required": ["query"],
28
+ },
29
+ )
30
+
31
+
32
+ def find_more_tools(query: str, corpus: Corpus, exclude: list[str] | None = None, k: int = 5) -> list[dict[str, str]]:
33
+ """Re-run resolution against the full corpus, skipping names already offered.
34
+ The caller executes this when the model calls `find_more_tools` and appends
35
+ the result as a normal tool result -- an append, never a rewrite."""
36
+ exclude = set(exclude or ())
37
+ found = resolve(query, corpus, k=k + len(exclude))
38
+ return [{"name": r.tool.name, "description": r.tool.description} for r in found if r.tool.name not in exclude][:k]