agentskills-retrieval 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentskills_retrieval-0.5.0/PKG-INFO +161 -0
- agentskills_retrieval-0.5.0/README.md +140 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/__init__.py +77 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/catalog.py +72 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/corpus.py +136 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/embedding.py +249 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/lexical.py +170 -0
- agentskills_retrieval-0.5.0/agentskills_retrieval/selector.py +87 -0
- agentskills_retrieval-0.5.0/pyproject.toml +26 -0
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agentskills-retrieval
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Query-time skill selection for the Agent Skills format (https://agentskills.io)
|
|
5
|
+
License: MIT
|
|
6
|
+
Author: Pratik Panda
|
|
7
|
+
Requires-Python: >=3.12,<4.0
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
16
|
+
Requires-Dist: agentskills-core (>=0.5.0,<1.0)
|
|
17
|
+
Project-URL: Homepage, https://agentskills.io
|
|
18
|
+
Project-URL: Repository, https://github.com/pratikxpanda/agentskills-sdk
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# agentskills-retrieval
|
|
22
|
+
|
|
23
|
+
[](https://pypi.org/project/agentskills-retrieval/)
|
|
24
|
+
[](https://github.com/pratikxpanda/agentskills-sdk/blob/main/LICENSE)
|
|
25
|
+
|
|
26
|
+
Query-time skill selection for the [Agent Skills](https://agentskills.io) SDK.
|
|
27
|
+
|
|
28
|
+
The skills catalog is injected into the system prompt on **every turn**, so its cost is linear in the number of registered skills. Two things get worse as a registry grows, not one: the token bill, and the accuracy of the model's choice. A fifty-skill registry is both more expensive *and* worse at picking the right skill.
|
|
29
|
+
|
|
30
|
+
`get_skills_catalog()` already accepts `include`, `exclude`, `tags` and `max_chars` — but every one of them requires the caller to know the answer in advance, and `max_chars` drops entries from the end, which is arbitrary with respect to relevance. This package narrows the catalog by *what was asked*.
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install agentskills-retrieval
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
There are no dependencies beyond `agentskills-core`. The default selector is pure Python.
|
|
39
|
+
|
|
40
|
+
## Usage
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
from agentskills_retrieval import LexicalSelector, build_selected_catalog
|
|
44
|
+
|
|
45
|
+
selector = LexicalSelector(registry)
|
|
46
|
+
catalog = await build_selected_catalog(
|
|
47
|
+
registry, selector, "the checkout API is returning 503s"
|
|
48
|
+
)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Or drive the two halves yourself, which is the whole integration surface:
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
selection = await selector.select("the checkout API is returning 503s", limit=5)
|
|
55
|
+
|
|
56
|
+
catalog = await registry.get_skills_catalog(
|
|
57
|
+
include=selection.skill_ids,
|
|
58
|
+
total=selection.considered,
|
|
59
|
+
)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`include=` is applied **before any metadata is fetched**, so narrowing fifty skills to five also avoids forty-five provider round trips. That matters most for exactly the registries this package exists for.
|
|
63
|
+
|
|
64
|
+
## Selectors
|
|
65
|
+
|
|
66
|
+
| Selector | Ranks by | Needs |
|
|
67
|
+
| --- | --- | --- |
|
|
68
|
+
| `LexicalSelector` | Okapi BM25 over name, description, `when_to_use` and tags | nothing |
|
|
69
|
+
| `EmbeddingSelector` | cosine similarity between query and skill vectors | an embedder you supply |
|
|
70
|
+
|
|
71
|
+
`LexicalSelector` is the default because it works the moment the package is installed. An embedding ranker is better at paraphrase — it can match "the site is down" to a skill that says "service degradation", which BM25 cannot, because they share no word — but it is also an API key, a network hop and a bill. A package whose only ranker needs all three is a package most people never switch on.
|
|
72
|
+
|
|
73
|
+
### Embedders
|
|
74
|
+
|
|
75
|
+
No embedding SDK is a dependency of anything here. The contract is one method:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
class OpenAIEmbedder:
|
|
79
|
+
embedder_id = "text-embedding-3-small"
|
|
80
|
+
|
|
81
|
+
async def embed(self, texts):
|
|
82
|
+
reply = await client.embeddings.create(model=self.embedder_id, input=list(texts))
|
|
83
|
+
return [item.embedding for item in reply.data]
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
from agentskills_retrieval import EmbeddingSelector
|
|
88
|
+
|
|
89
|
+
selector = EmbeddingSelector(registry, OpenAIEmbedder())
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`load_embedder("myapp.embedders:build")` resolves the same thing from a dotted path when the name comes from configuration, mirroring how the eval harness resolves chat models.
|
|
93
|
+
|
|
94
|
+
### Caching
|
|
95
|
+
|
|
96
|
+
Skill vectors are cached by content hash, so re-registering an unchanged skill is free and editing one invalidates only itself. `embedder_id` is part of every cache key, because two models' vectors are not comparable and a cache that mixes them silently returns nonsense.
|
|
97
|
+
|
|
98
|
+
The default cache is an in-process dict. Persistence is a two-method protocol — a hosted registry should not re-embed its corpus every time a process starts:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
class RedisEmbeddingCache:
|
|
102
|
+
def get(self, key: str) -> list[float] | None: ...
|
|
103
|
+
def set(self, key: str, vector: list[float]) -> None: ...
|
|
104
|
+
|
|
105
|
+
selector = EmbeddingSelector(registry, embedder, cache=RedisEmbeddingCache())
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## `when_not_to_use` is a penalty, not a match
|
|
109
|
+
|
|
110
|
+
Selection metadata is indexed in two halves. `description`, `when_to_use`, the skill name and its tags are evidence *for* a skill. `when_not_to_use` is evidence *against* it, and is scored separately and subtracted.
|
|
111
|
+
|
|
112
|
+
Folding both into one bag of words would make a skill match the very query its author wrote it to disclaim — "not for local test failures" would make the skill *more* likely to win on a query about a local test failure, because the words line up. The weight is `0.5` rather than `1.0` because a disclaimer is weaker evidence than a description: authors write far fewer of them, and phrase them loosely. Pass `negative_weight=0.0` to ignore them.
|
|
113
|
+
|
|
114
|
+
## Selection is visible or it is not debuggable
|
|
115
|
+
|
|
116
|
+
From inside an agent, a skill that was ranked out is indistinguishable from a skill that was never registered. So:
|
|
117
|
+
|
|
118
|
+
- Every selection is logged at `INFO` on `agentskills.retrieval.*` with the scores that produced it.
|
|
119
|
+
- `Selection.rejected` carries what did **not** make the cut, also best-first. "The right skill scored just under the floor" and "the right skill was never registered" are different bugs that look identical without it.
|
|
120
|
+
- `build_selected_catalog` passes `total=` so the catalog reports the shortfall itself — `shown`/`total` on the XML root, a closing note in Markdown.
|
|
121
|
+
|
|
122
|
+
## Floors, and what they can and cannot catch
|
|
123
|
+
|
|
124
|
+
Returning the five best of fifty irrelevant skills is worse than returning nothing, so both selectors take a `min_score`.
|
|
125
|
+
|
|
126
|
+
- **Embeddings**: cosine is bounded and comparable across corpora, so the floor is meaningful. The `0.25` default is still model-specific — some embedders put unrelated text around 0.7 — so treat it as a starting point.
|
|
127
|
+
- **BM25**: scores are unbounded and corpus-relative, so there is no meaningful absolute floor above zero. The default of `0.0` catches the case that matters — the query shares no term with any skill — but it **cannot** catch a query that matches a common word and is nonetheless irrelevant. That limit is real, and it is why the recall numbers below are measured rather than assumed.
|
|
128
|
+
|
|
129
|
+
When nothing clears the floor, `build_selected_catalog` returns the **full** catalog. "Selection has no opinion" is not "the agent should have no skills": a wrong prune silently removes a capability, which is a worse failure than a few wasted tokens. The fallback is logged.
|
|
130
|
+
|
|
131
|
+
## What the query should be
|
|
132
|
+
|
|
133
|
+
`select()` takes a string and never derives one, because the obvious default is wrong. The last user message alone fails for any conversation where the topic was established several turns ago — "try that again" ranks against nothing. Concatenating the whole history fails the other way, dragging in every topic the conversation has touched. The caller knows its own conversation shape; this package does not, and guessing on its behalf would be a silent accuracy regression rather than an obvious one.
|
|
134
|
+
|
|
135
|
+
## Measured recall
|
|
136
|
+
|
|
137
|
+
A ranker shipped without a measurement is a guess with an API. The fixture set in [`tests/conftest.py`](https://github.com/pratikxpanda/agentskills-sdk/blob/main/packages/retrieval/agentskills-retrieval/tests/conftest.py) pairs realistic queries with the skill that should win, and [`tests/test_recall.py`](https://github.com/pratikxpanda/agentskills-sdk/blob/main/packages/retrieval/agentskills-retrieval/tests/test_recall.py) asserts a floor that CI enforces, so a change that makes ranking worse fails the build.
|
|
138
|
+
|
|
139
|
+
`LexicalSelector` over the fixture corpus:
|
|
140
|
+
|
|
141
|
+
| Metric | Score |
|
|
142
|
+
| --- | --- |
|
|
143
|
+
| recall@1 | 0.85 |
|
|
144
|
+
| recall@3 | 1.00 |
|
|
145
|
+
|
|
146
|
+
The `EmbeddingSelector` figures are measured against a deterministic hashing embedder, which tests the plumbing rather than any real model's quality; a number from a stub embedder would be a claim about nothing. Run the harness against your own embedder before trusting it in production.
|
|
147
|
+
|
|
148
|
+
Both numbers are over a small synthetic corpus. They are a regression guard, not a benchmark.
|
|
149
|
+
|
|
150
|
+
## Not a default
|
|
151
|
+
|
|
152
|
+
Nothing here is enabled unless you enable it. A registry that is not asked to select behaves exactly as it did before this package existed, byte for byte.
|
|
153
|
+
|
|
154
|
+
## Security
|
|
155
|
+
|
|
156
|
+
Selection reads skill metadata only, never bodies or resources. It executes nothing. An embedder you supply is your own code and your own network egress; this package neither imports nor configures one.
|
|
157
|
+
|
|
158
|
+
## License
|
|
159
|
+
|
|
160
|
+
MIT
|
|
161
|
+
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
# agentskills-retrieval
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/agentskills-retrieval/)
|
|
4
|
+
[](https://github.com/pratikxpanda/agentskills-sdk/blob/main/LICENSE)
|
|
5
|
+
|
|
6
|
+
Query-time skill selection for the [Agent Skills](https://agentskills.io) SDK.
|
|
7
|
+
|
|
8
|
+
The skills catalog is injected into the system prompt on **every turn**, so its cost is linear in the number of registered skills. Two things get worse as a registry grows, not one: the token bill, and the accuracy of the model's choice. A fifty-skill registry is both more expensive *and* worse at picking the right skill.
|
|
9
|
+
|
|
10
|
+
`get_skills_catalog()` already accepts `include`, `exclude`, `tags` and `max_chars` — but every one of them requires the caller to know the answer in advance, and `max_chars` drops entries from the end, which is arbitrary with respect to relevance. This package narrows the catalog by *what was asked*.
|
|
11
|
+
|
|
12
|
+
## Installation
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
pip install agentskills-retrieval
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
There are no dependencies beyond `agentskills-core`. The default selector is pure Python.
|
|
19
|
+
|
|
20
|
+
## Usage
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
from agentskills_retrieval import LexicalSelector, build_selected_catalog
|
|
24
|
+
|
|
25
|
+
selector = LexicalSelector(registry)
|
|
26
|
+
catalog = await build_selected_catalog(
|
|
27
|
+
registry, selector, "the checkout API is returning 503s"
|
|
28
|
+
)
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Or drive the two halves yourself, which is the whole integration surface:
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
selection = await selector.select("the checkout API is returning 503s", limit=5)
|
|
35
|
+
|
|
36
|
+
catalog = await registry.get_skills_catalog(
|
|
37
|
+
include=selection.skill_ids,
|
|
38
|
+
total=selection.considered,
|
|
39
|
+
)
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`include=` is applied **before any metadata is fetched**, so narrowing fifty skills to five also avoids forty-five provider round trips. That matters most for exactly the registries this package exists for.
|
|
43
|
+
|
|
44
|
+
## Selectors
|
|
45
|
+
|
|
46
|
+
| Selector | Ranks by | Needs |
|
|
47
|
+
| --- | --- | --- |
|
|
48
|
+
| `LexicalSelector` | Okapi BM25 over name, description, `when_to_use` and tags | nothing |
|
|
49
|
+
| `EmbeddingSelector` | cosine similarity between query and skill vectors | an embedder you supply |
|
|
50
|
+
|
|
51
|
+
`LexicalSelector` is the default because it works the moment the package is installed. An embedding ranker is better at paraphrase — it can match "the site is down" to a skill that says "service degradation", which BM25 cannot, because they share no word — but it is also an API key, a network hop and a bill. A package whose only ranker needs all three is a package most people never switch on.
|
|
52
|
+
|
|
53
|
+
### Embedders
|
|
54
|
+
|
|
55
|
+
No embedding SDK is a dependency of anything here. The contract is one method:
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
class OpenAIEmbedder:
|
|
59
|
+
embedder_id = "text-embedding-3-small"
|
|
60
|
+
|
|
61
|
+
async def embed(self, texts):
|
|
62
|
+
reply = await client.embeddings.create(model=self.embedder_id, input=list(texts))
|
|
63
|
+
return [item.embedding for item in reply.data]
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from agentskills_retrieval import EmbeddingSelector
|
|
68
|
+
|
|
69
|
+
selector = EmbeddingSelector(registry, OpenAIEmbedder())
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`load_embedder("myapp.embedders:build")` resolves the same thing from a dotted path when the name comes from configuration, mirroring how the eval harness resolves chat models.
|
|
73
|
+
|
|
74
|
+
### Caching
|
|
75
|
+
|
|
76
|
+
Skill vectors are cached by content hash, so re-registering an unchanged skill is free and editing one invalidates only itself. `embedder_id` is part of every cache key, because two models' vectors are not comparable and a cache that mixes them silently returns nonsense.
|
|
77
|
+
|
|
78
|
+
The default cache is an in-process dict. Persistence is a two-method protocol — a hosted registry should not re-embed its corpus every time a process starts:
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
class RedisEmbeddingCache:
|
|
82
|
+
def get(self, key: str) -> list[float] | None: ...
|
|
83
|
+
def set(self, key: str, vector: list[float]) -> None: ...
|
|
84
|
+
|
|
85
|
+
selector = EmbeddingSelector(registry, embedder, cache=RedisEmbeddingCache())
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## `when_not_to_use` is a penalty, not a match
|
|
89
|
+
|
|
90
|
+
Selection metadata is indexed in two halves. `description`, `when_to_use`, the skill name and its tags are evidence *for* a skill. `when_not_to_use` is evidence *against* it, and is scored separately and subtracted.
|
|
91
|
+
|
|
92
|
+
Folding both into one bag of words would make a skill match the very query its author wrote it to disclaim — "not for local test failures" would make the skill *more* likely to win on a query about a local test failure, because the words line up. The weight is `0.5` rather than `1.0` because a disclaimer is weaker evidence than a description: authors write far fewer of them, and phrase them loosely. Pass `negative_weight=0.0` to ignore them.
|
|
93
|
+
|
|
94
|
+
## Selection is visible or it is not debuggable
|
|
95
|
+
|
|
96
|
+
From inside an agent, a skill that was ranked out is indistinguishable from a skill that was never registered. So:
|
|
97
|
+
|
|
98
|
+
- Every selection is logged at `INFO` on `agentskills.retrieval.*` with the scores that produced it.
|
|
99
|
+
- `Selection.rejected` carries what did **not** make the cut, also best-first. "The right skill scored just under the floor" and "the right skill was never registered" are different bugs that look identical without it.
|
|
100
|
+
- `build_selected_catalog` passes `total=` so the catalog reports the shortfall itself — `shown`/`total` on the XML root, a closing note in Markdown.
|
|
101
|
+
|
|
102
|
+
## Floors, and what they can and cannot catch
|
|
103
|
+
|
|
104
|
+
Returning the five best of fifty irrelevant skills is worse than returning nothing, so both selectors take a `min_score`.
|
|
105
|
+
|
|
106
|
+
- **Embeddings**: cosine is bounded and comparable across corpora, so the floor is meaningful. The `0.25` default is still model-specific — some embedders put unrelated text around 0.7 — so treat it as a starting point.
|
|
107
|
+
- **BM25**: scores are unbounded and corpus-relative, so there is no meaningful absolute floor above zero. The default of `0.0` catches the case that matters — the query shares no term with any skill — but it **cannot** catch a query that matches a common word and is nonetheless irrelevant. That limit is real, and it is why the recall numbers below are measured rather than assumed.
|
|
108
|
+
|
|
109
|
+
When nothing clears the floor, `build_selected_catalog` returns the **full** catalog. "Selection has no opinion" is not "the agent should have no skills": a wrong prune silently removes a capability, which is a worse failure than a few wasted tokens. The fallback is logged.
|
|
110
|
+
|
|
111
|
+
## What the query should be
|
|
112
|
+
|
|
113
|
+
`select()` takes a string and never derives one, because the obvious default is wrong. The last user message alone fails for any conversation where the topic was established several turns ago — "try that again" ranks against nothing. Concatenating the whole history fails the other way, dragging in every topic the conversation has touched. The caller knows its own conversation shape; this package does not, and guessing on its behalf would be a silent accuracy regression rather than an obvious one.
|
|
114
|
+
|
|
115
|
+
## Measured recall
|
|
116
|
+
|
|
117
|
+
A ranker shipped without a measurement is a guess with an API. The fixture set in [`tests/conftest.py`](https://github.com/pratikxpanda/agentskills-sdk/blob/main/packages/retrieval/agentskills-retrieval/tests/conftest.py) pairs realistic queries with the skill that should win, and [`tests/test_recall.py`](https://github.com/pratikxpanda/agentskills-sdk/blob/main/packages/retrieval/agentskills-retrieval/tests/test_recall.py) asserts a floor that CI enforces, so a change that makes ranking worse fails the build.
|
|
118
|
+
|
|
119
|
+
`LexicalSelector` over the fixture corpus:
|
|
120
|
+
|
|
121
|
+
| Metric | Score |
|
|
122
|
+
| --- | --- |
|
|
123
|
+
| recall@1 | 0.85 |
|
|
124
|
+
| recall@3 | 1.00 |
|
|
125
|
+
|
|
126
|
+
The `EmbeddingSelector` figures are measured against a deterministic hashing embedder, which tests the plumbing rather than any real model's quality; a number from a stub embedder would be a claim about nothing. Run the harness against your own embedder before trusting it in production.
|
|
127
|
+
|
|
128
|
+
Both numbers are over a small synthetic corpus. They are a regression guard, not a benchmark.
|
|
129
|
+
|
|
130
|
+
## Not a default
|
|
131
|
+
|
|
132
|
+
Nothing here is enabled unless you enable it. A registry that is not asked to select behaves exactly as it did before this package existed, byte for byte.
|
|
133
|
+
|
|
134
|
+
## Security
|
|
135
|
+
|
|
136
|
+
Selection reads skill metadata only, never bodies or resources. It executes nothing. An embedder you supply is your own code and your own network egress; this package neither imports nor configures one.
|
|
137
|
+
|
|
138
|
+
## License
|
|
139
|
+
|
|
140
|
+
MIT
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Query-time skill selection for the Agent Skills SDK.
|
|
2
|
+
|
|
3
|
+
The catalog is injected on every turn, so per-turn prompt cost is
|
|
4
|
+
linear in the number of registered skills — and selection accuracy
|
|
5
|
+
falls as the candidate list grows, so a large registry is both more
|
|
6
|
+
expensive and worse at choosing. This package narrows the catalog to
|
|
7
|
+
what the current query is actually about.
|
|
8
|
+
|
|
9
|
+
Two selectors ship. :class:`LexicalSelector` is BM25 and needs nothing
|
|
10
|
+
installed; :class:`EmbeddingSelector` takes any embedder behind a
|
|
11
|
+
one-method protocol. Both return scores, and both compose with the
|
|
12
|
+
``include=`` filter the registry already has::
|
|
13
|
+
|
|
14
|
+
from agentskills_retrieval import LexicalSelector, build_selected_catalog
|
|
15
|
+
|
|
16
|
+
selector = LexicalSelector(registry)
|
|
17
|
+
catalog = await build_selected_catalog(registry, selector, "checkout is down")
|
|
18
|
+
|
|
19
|
+
Nothing here is on by default. A registry that is not asked to select
|
|
20
|
+
behaves exactly as it did before this package existed.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from agentskills_retrieval.catalog import build_selected_catalog
|
|
24
|
+
from agentskills_retrieval.corpus import (
|
|
25
|
+
STOPWORDS,
|
|
26
|
+
SkillDocument,
|
|
27
|
+
build_corpus,
|
|
28
|
+
document_of,
|
|
29
|
+
tokenize,
|
|
30
|
+
)
|
|
31
|
+
from agentskills_retrieval.embedding import (
|
|
32
|
+
DEFAULT_MIN_SCORE as DEFAULT_EMBEDDING_MIN_SCORE,
|
|
33
|
+
)
|
|
34
|
+
from agentskills_retrieval.embedding import (
|
|
35
|
+
Embedder,
|
|
36
|
+
EmbeddingCache,
|
|
37
|
+
EmbeddingSelector,
|
|
38
|
+
InMemoryEmbeddingCache,
|
|
39
|
+
cosine,
|
|
40
|
+
load_embedder,
|
|
41
|
+
)
|
|
42
|
+
from agentskills_retrieval.lexical import (
|
|
43
|
+
DEFAULT_MIN_SCORE as DEFAULT_LEXICAL_MIN_SCORE,
|
|
44
|
+
)
|
|
45
|
+
from agentskills_retrieval.lexical import (
|
|
46
|
+
NEGATIVE_WEIGHT,
|
|
47
|
+
LexicalSelector,
|
|
48
|
+
)
|
|
49
|
+
from agentskills_retrieval.selector import (
|
|
50
|
+
DEFAULT_LIMIT,
|
|
51
|
+
ScoredSkill,
|
|
52
|
+
Selection,
|
|
53
|
+
SkillSelector,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
__all__ = [
|
|
57
|
+
"DEFAULT_EMBEDDING_MIN_SCORE",
|
|
58
|
+
"DEFAULT_LEXICAL_MIN_SCORE",
|
|
59
|
+
"DEFAULT_LIMIT",
|
|
60
|
+
"NEGATIVE_WEIGHT",
|
|
61
|
+
"STOPWORDS",
|
|
62
|
+
"Embedder",
|
|
63
|
+
"EmbeddingCache",
|
|
64
|
+
"EmbeddingSelector",
|
|
65
|
+
"InMemoryEmbeddingCache",
|
|
66
|
+
"LexicalSelector",
|
|
67
|
+
"ScoredSkill",
|
|
68
|
+
"Selection",
|
|
69
|
+
"SkillDocument",
|
|
70
|
+
"SkillSelector",
|
|
71
|
+
"build_corpus",
|
|
72
|
+
"build_selected_catalog",
|
|
73
|
+
"cosine",
|
|
74
|
+
"document_of",
|
|
75
|
+
"load_embedder",
|
|
76
|
+
"tokenize",
|
|
77
|
+
]
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Turning a selection into a catalog.
|
|
2
|
+
|
|
3
|
+
The one function here is the whole integration surface: rank, then
|
|
4
|
+
hand the winning IDs to the filter ``get_skills_catalog`` already has.
|
|
5
|
+
Core learns nothing about ranking; retrieval learns nothing about
|
|
6
|
+
rendering.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import TYPE_CHECKING, Any, Literal
|
|
12
|
+
|
|
13
|
+
from agentskills_core import get_logger
|
|
14
|
+
from agentskills_retrieval.selector import DEFAULT_LIMIT
|
|
15
|
+
|
|
16
|
+
if TYPE_CHECKING:
|
|
17
|
+
from agentskills_core import SkillRegistry
|
|
18
|
+
from agentskills_retrieval.selector import SkillSelector
|
|
19
|
+
|
|
20
|
+
_logger = get_logger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
async def build_selected_catalog(
|
|
24
|
+
registry: SkillRegistry,
|
|
25
|
+
selector: SkillSelector,
|
|
26
|
+
query: str,
|
|
27
|
+
*,
|
|
28
|
+
limit: int = DEFAULT_LIMIT,
|
|
29
|
+
format: Literal["xml", "markdown"] = "xml",
|
|
30
|
+
**catalog_kwargs: Any,
|
|
31
|
+
) -> str:
|
|
32
|
+
"""Rank the registry against *query* and build a catalog of the winners.
|
|
33
|
+
|
|
34
|
+
The result reports its own narrowing — ``shown``/``total`` on the
|
|
35
|
+
XML root, a closing note in Markdown — because from inside an agent
|
|
36
|
+
a skill that was ranked out is indistinguishable from one that was
|
|
37
|
+
never registered, and that is a miserable thing to debug.
|
|
38
|
+
|
|
39
|
+
When nothing clears the selector's floor the **full** catalog is
|
|
40
|
+
returned. "Selection has no opinion" is not "the agent should have
|
|
41
|
+
no skills": a wrong prune silently removes a capability, which is a
|
|
42
|
+
worse failure than a few wasted tokens. The fallback is logged.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
registry: The registry to build from.
|
|
46
|
+
selector: Any :class:`~agentskills_retrieval.SkillSelector`.
|
|
47
|
+
query: Text to rank against.
|
|
48
|
+
limit: Maximum number of skills to advertise.
|
|
49
|
+
format: ``"xml"`` or ``"markdown"``, as for
|
|
50
|
+
:meth:`~agentskills_core.SkillRegistry.get_skills_catalog`.
|
|
51
|
+
**catalog_kwargs: Passed straight through to
|
|
52
|
+
:meth:`~agentskills_core.SkillRegistry.get_skills_catalog`.
|
|
53
|
+
``include`` and ``total`` are supplied by this function.
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
A catalog string ready for a system prompt.
|
|
57
|
+
"""
|
|
58
|
+
selection = await selector.select(query, limit=limit)
|
|
59
|
+
if selection.is_empty:
|
|
60
|
+
_logger.info(
|
|
61
|
+
"Selection matched nothing for %r; falling back to the full catalog of %d skills",
|
|
62
|
+
query,
|
|
63
|
+
selection.considered,
|
|
64
|
+
)
|
|
65
|
+
return await registry.get_skills_catalog(format=format, **catalog_kwargs)
|
|
66
|
+
|
|
67
|
+
return await registry.get_skills_catalog(
|
|
68
|
+
format=format,
|
|
69
|
+
include=selection.skill_ids,
|
|
70
|
+
total=selection.considered,
|
|
71
|
+
**catalog_kwargs,
|
|
72
|
+
)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""The searchable text of a skill, and how it is tokenized.
|
|
2
|
+
|
|
3
|
+
Both selectors rank the same corpus: whatever the catalog would have
|
|
4
|
+
shown a model. Building it here rather than in each selector means a
|
|
5
|
+
lexical and an embedding ranking are answering the same question over
|
|
6
|
+
the same words, so a difference between them is a difference in method
|
|
7
|
+
and not in what they were allowed to read.
|
|
8
|
+
|
|
9
|
+
``when_not_to_use`` is deliberately kept apart from the rest. It is
|
|
10
|
+
evidence *against* a skill, and folding it into one bag of words would
|
|
11
|
+
make a skill match the very query its author wrote it to disclaim.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import asyncio
|
|
17
|
+
import re
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from hashlib import blake2b
|
|
20
|
+
from typing import TYPE_CHECKING, Any
|
|
21
|
+
|
|
22
|
+
from agentskills_core import SELECTION_FIELDS, get_logger
|
|
23
|
+
|
|
24
|
+
if TYPE_CHECKING:
|
|
25
|
+
from agentskills_core import Skill, SkillRegistry
|
|
26
|
+
|
|
27
|
+
_logger = get_logger(__name__)
|
|
28
|
+
|
|
29
|
+
_WORD = re.compile(r"[a-z0-9]+")
|
|
30
|
+
|
|
31
|
+
#: Words carried by almost every skill description, so they separate nothing.
|
|
32
|
+
#:
|
|
33
|
+
#: Deliberately tiny. A long list is a language model of its own that
|
|
34
|
+
#: nobody here is qualified to maintain, and BM25 already discounts a
|
|
35
|
+
#: term that appears in most documents. This exists only to stop very
|
|
36
|
+
#: short queries being dominated by their function words.
|
|
37
|
+
STOPWORDS = frozenset(
|
|
38
|
+
[
|
|
39
|
+
"a", "an", "and", "are", "as", "at", "be", "by", "for", "from",
|
|
40
|
+
"how", "in", "into", "is", "it", "of", "on", "or", "that", "the",
|
|
41
|
+
"this", "to", "use", "used", "using", "was", "what", "when",
|
|
42
|
+
"where", "which", "who", "why", "with", "you", "your",
|
|
43
|
+
]
|
|
44
|
+
) # fmt: skip
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def tokenize(text: str) -> list[str]:
|
|
48
|
+
"""Split *text* into lowercase alphanumeric terms, minus stopwords.
|
|
49
|
+
|
|
50
|
+
No stemming: it would need a dependency, and this package's whole
|
|
51
|
+
claim is that it is useful with none. The cost is real — "deploys"
|
|
52
|
+
will not match "deploy" — and it is the main thing an embedding
|
|
53
|
+
selector buys back.
|
|
54
|
+
"""
|
|
55
|
+
return [word for word in _WORD.findall(text.casefold()) if word not in STOPWORDS]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True)
|
|
59
|
+
class SkillDocument:
|
|
60
|
+
"""One skill, as a ranker sees it.
|
|
61
|
+
|
|
62
|
+
Attributes:
|
|
63
|
+
skill_id: The registered ID, which is what a selection returns.
|
|
64
|
+
positive_text: Name, description, ``when_to_use`` and tags.
|
|
65
|
+
negative_text: ``when_not_to_use``, scored separately.
|
|
66
|
+
content_hash: Digest of the text above, so an embedding can be
|
|
67
|
+
cached against it and invalidated by nothing else.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
skill_id: str
|
|
71
|
+
positive_text: str
|
|
72
|
+
negative_text: str
|
|
73
|
+
content_hash: str
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def positive_terms(self) -> list[str]:
|
|
77
|
+
"""The tokens a query is matched against."""
|
|
78
|
+
return tokenize(self.positive_text)
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def negative_terms(self) -> list[str]:
|
|
82
|
+
"""The tokens a query is penalised for matching."""
|
|
83
|
+
return tokenize(self.negative_text)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _strings(value: Any) -> list[str]:
|
|
87
|
+
"""Return *value* as a list of non-empty strings, or nothing."""
|
|
88
|
+
if not isinstance(value, list):
|
|
89
|
+
return []
|
|
90
|
+
return [item.strip() for item in value if isinstance(item, str) and item.strip()]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def document_of(skill_id: str, meta: dict[str, Any]) -> SkillDocument:
|
|
94
|
+
"""Build the searchable document for one skill's metadata."""
|
|
95
|
+
positive = [
|
|
96
|
+
str(meta.get("name") or skill_id),
|
|
97
|
+
str(meta.get("description") or ""),
|
|
98
|
+
*_strings(meta.get(SELECTION_FIELDS[0])),
|
|
99
|
+
]
|
|
100
|
+
container = meta.get("metadata")
|
|
101
|
+
if isinstance(container, dict):
|
|
102
|
+
positive += _strings(container.get("tags"))
|
|
103
|
+
|
|
104
|
+
negative = _strings(meta.get(SELECTION_FIELDS[1]))
|
|
105
|
+
|
|
106
|
+
positive_text = "\n".join(part for part in positive if part)
|
|
107
|
+
negative_text = "\n".join(negative)
|
|
108
|
+
digest = blake2b(f"{positive_text}\x00{negative_text}".encode(), digest_size=16)
|
|
109
|
+
return SkillDocument(
|
|
110
|
+
skill_id=skill_id,
|
|
111
|
+
positive_text=positive_text,
|
|
112
|
+
negative_text=negative_text,
|
|
113
|
+
content_hash=digest.hexdigest(),
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
async def build_corpus(registry: SkillRegistry, *, concurrency: int = 8) -> list[SkillDocument]:
|
|
118
|
+
"""Fetch metadata for every registered skill and index it.
|
|
119
|
+
|
|
120
|
+
Args:
|
|
121
|
+
registry: The registry to read.
|
|
122
|
+
concurrency: Ceiling on simultaneous metadata fetches, which
|
|
123
|
+
matters when the provider is network-backed.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
One :class:`SkillDocument` per registered skill, in ID order.
|
|
127
|
+
"""
|
|
128
|
+
semaphore = asyncio.Semaphore(concurrency)
|
|
129
|
+
|
|
130
|
+
async def fetch(skill: Skill) -> SkillDocument:
|
|
131
|
+
async with semaphore:
|
|
132
|
+
return document_of(skill.get_id(), await skill.get_metadata())
|
|
133
|
+
|
|
134
|
+
corpus = list(await asyncio.gather(*(fetch(skill) for skill in registry.list_skills())))
|
|
135
|
+
_logger.debug("Indexed %d skills for selection", len(corpus))
|
|
136
|
+
return corpus
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""Embedding-based selection, with no vendor SDK anywhere in the tree.
|
|
2
|
+
|
|
3
|
+
BM25 cannot match "the site is down" to a skill that says "service
|
|
4
|
+
degradation", because they share no word. Embeddings can, and the
|
|
5
|
+
price is a model — so the model stays outside the package, behind a
|
|
6
|
+
one-method protocol resolved from a dotted path at run time, exactly
|
|
7
|
+
as the eval harness resolves chat models. Nothing here imports
|
|
8
|
+
anything a user did not ask for.
|
|
9
|
+
|
|
10
|
+
Vectors are cached by content hash, so re-registering an unchanged
|
|
11
|
+
skill costs nothing and editing one invalidates only itself. The cache
|
|
12
|
+
is a protocol too: a hosted registry should not re-embed its corpus
|
|
13
|
+
every time a process starts.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import math
|
|
19
|
+
from importlib import import_module
|
|
20
|
+
from typing import TYPE_CHECKING, Protocol, runtime_checkable
|
|
21
|
+
|
|
22
|
+
from agentskills_core import get_logger
|
|
23
|
+
from agentskills_retrieval.corpus import SkillDocument, build_corpus
|
|
24
|
+
from agentskills_retrieval.selector import DEFAULT_LIMIT, ScoredSkill, Selection
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from collections.abc import Sequence
|
|
28
|
+
|
|
29
|
+
from agentskills_core import SkillRegistry
|
|
30
|
+
|
|
31
|
+
_logger = get_logger(__name__)
|
|
32
|
+
|
|
33
|
+
#: Cosine similarity below this is treated as no match.
|
|
34
|
+
#:
|
|
35
|
+
#: Unlike BM25 this floor is meaningful, because cosine is bounded and
|
|
36
|
+
#: comparable across corpora. It is still model-specific — some
|
|
37
|
+
#: embedders put unrelated text around 0.7 — so it is a constructor
|
|
38
|
+
#: argument and this is only a starting point.
|
|
39
|
+
DEFAULT_MIN_SCORE = 0.25
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@runtime_checkable
|
|
43
|
+
class Embedder(Protocol):
|
|
44
|
+
"""The whole contract: a name, and a way to turn text into vectors.
|
|
45
|
+
|
|
46
|
+
Implement it over any client::
|
|
47
|
+
|
|
48
|
+
class OpenAIEmbedder:
|
|
49
|
+
embedder_id = "text-embedding-3-small"
|
|
50
|
+
|
|
51
|
+
async def embed(self, texts):
|
|
52
|
+
reply = await client.embeddings.create(
|
|
53
|
+
model=self.embedder_id, input=list(texts)
|
|
54
|
+
)
|
|
55
|
+
return [item.embedding for item in reply.data]
|
|
56
|
+
|
|
57
|
+
``embedder_id`` is part of every cache key, because two models'
|
|
58
|
+
vectors are not comparable and a cache that mixes them silently
|
|
59
|
+
returns nonsense.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
embedder_id: str
|
|
63
|
+
|
|
64
|
+
async def embed(self, texts: Sequence[str]) -> list[list[float]]:
|
|
65
|
+
"""Return one vector per text, in the same order."""
|
|
66
|
+
...
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def load_embedder(spec: str) -> Embedder:
|
|
70
|
+
"""Resolve ``module:factory`` to an embedder.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
spec: A dotted module path, a colon, and the name of a
|
|
74
|
+
zero-argument callable returning an :class:`Embedder`.
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
The embedder the factory built.
|
|
78
|
+
|
|
79
|
+
Raises:
|
|
80
|
+
ValueError: If the spec is malformed, cannot be imported, or
|
|
81
|
+
does not produce something with ``embedder_id`` and
|
|
82
|
+
``embed``.
|
|
83
|
+
"""
|
|
84
|
+
module_name, _, factory_name = spec.partition(":")
|
|
85
|
+
if not module_name or not factory_name:
|
|
86
|
+
raise ValueError(f"embedder spec must look like 'module:factory', got {spec!r}")
|
|
87
|
+
|
|
88
|
+
try:
|
|
89
|
+
module = import_module(module_name)
|
|
90
|
+
except ImportError as exc:
|
|
91
|
+
raise ValueError(f"cannot import '{module_name}': {exc}") from exc
|
|
92
|
+
|
|
93
|
+
try:
|
|
94
|
+
factory = getattr(module, factory_name)
|
|
95
|
+
except AttributeError as exc:
|
|
96
|
+
raise ValueError(f"'{module_name}' has no attribute '{factory_name}'") from exc
|
|
97
|
+
|
|
98
|
+
embedder = factory()
|
|
99
|
+
if not isinstance(embedder, Embedder):
|
|
100
|
+
raise ValueError(
|
|
101
|
+
f"{spec} returned {type(embedder).__name__}, which has no 'embedder_id' and 'embed'."
|
|
102
|
+
)
|
|
103
|
+
return embedder
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@runtime_checkable
|
|
107
|
+
class EmbeddingCache(Protocol):
|
|
108
|
+
"""Somewhere to keep vectors between processes.
|
|
109
|
+
|
|
110
|
+
Two methods, both synchronous, because the interesting backends
|
|
111
|
+
(a file, a table, Redis) are all fast enough that making callers
|
|
112
|
+
write an async adapter buys nothing.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
def get(self, key: str) -> list[float] | None:
|
|
116
|
+
"""Return the vector stored under *key*, or ``None``."""
|
|
117
|
+
...
|
|
118
|
+
|
|
119
|
+
def set(self, key: str, vector: list[float]) -> None:
|
|
120
|
+
"""Store *vector* under *key*."""
|
|
121
|
+
...
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class InMemoryEmbeddingCache:
|
|
125
|
+
"""The default cache: a dict that dies with the process."""
|
|
126
|
+
|
|
127
|
+
def __init__(self) -> None:
|
|
128
|
+
self._vectors: dict[str, list[float]] = {}
|
|
129
|
+
|
|
130
|
+
def get(self, key: str) -> list[float] | None:
|
|
131
|
+
"""Return the vector stored under *key*, or ``None``."""
|
|
132
|
+
return self._vectors.get(key)
|
|
133
|
+
|
|
134
|
+
def set(self, key: str, vector: list[float]) -> None:
|
|
135
|
+
"""Store *vector* under *key*."""
|
|
136
|
+
self._vectors[key] = vector
|
|
137
|
+
|
|
138
|
+
def __len__(self) -> int:
|
|
139
|
+
"""Number of cached vectors, which is what tests assert on."""
|
|
140
|
+
return len(self._vectors)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def cosine(left: Sequence[float], right: Sequence[float]) -> float:
|
|
144
|
+
"""Return the cosine similarity of two vectors, or 0.0 if either is zero."""
|
|
145
|
+
dot = sum(a * b for a, b in zip(left, right, strict=True))
|
|
146
|
+
norm = math.sqrt(sum(a * a for a in left)) * math.sqrt(sum(b * b for b in right))
|
|
147
|
+
return dot / norm if norm else 0.0
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class EmbeddingSelector:
|
|
151
|
+
"""Select skills by cosine similarity between query and skill vectors.
|
|
152
|
+
|
|
153
|
+
Args:
|
|
154
|
+
registry: The registry whose skills are ranked.
|
|
155
|
+
embedder: Anything satisfying :class:`Embedder`. Build one with
|
|
156
|
+
:func:`load_embedder` when the name comes from config.
|
|
157
|
+
cache: Where skill vectors are kept. Defaults to an in-process
|
|
158
|
+
dict; pass your own to survive a restart.
|
|
159
|
+
min_score: Cosine below this is rejected. See
|
|
160
|
+
:data:`DEFAULT_MIN_SCORE`.
|
|
161
|
+
negative_weight: How much similarity to ``when_not_to_use``
|
|
162
|
+
subtracts. Pass ``0.0`` to ignore disclaimers.
|
|
163
|
+
"""
|
|
164
|
+
|
|
165
|
+
def __init__(
|
|
166
|
+
self,
|
|
167
|
+
registry: SkillRegistry,
|
|
168
|
+
embedder: Embedder,
|
|
169
|
+
*,
|
|
170
|
+
cache: EmbeddingCache | None = None,
|
|
171
|
+
min_score: float = DEFAULT_MIN_SCORE,
|
|
172
|
+
negative_weight: float = 0.5,
|
|
173
|
+
) -> None:
|
|
174
|
+
self._registry = registry
|
|
175
|
+
self._embedder = embedder
|
|
176
|
+
self._cache: EmbeddingCache = cache if cache is not None else InMemoryEmbeddingCache()
|
|
177
|
+
self._min_score = min_score
|
|
178
|
+
self._negative_weight = negative_weight
|
|
179
|
+
self._corpus: list[SkillDocument] = []
|
|
180
|
+
self._positive: list[list[float]] = []
|
|
181
|
+
self._negative: list[list[float] | None] = []
|
|
182
|
+
self._indexed_ids: tuple[str, ...] = ()
|
|
183
|
+
|
|
184
|
+
def _key(self, content_hash: str, field: str) -> str:
|
|
185
|
+
return f"{self._embedder.embedder_id}:{field}:{content_hash}"
|
|
186
|
+
|
|
187
|
+
async def _vectors(self, wanted: list[tuple[str, str]]) -> list[list[float]]:
|
|
188
|
+
"""Return a vector per ``(cache key, text)``, embedding only misses."""
|
|
189
|
+
cached = [self._cache.get(key) for key, _ in wanted]
|
|
190
|
+
missing = [i for i, vector in enumerate(cached) if vector is None]
|
|
191
|
+
if missing:
|
|
192
|
+
fresh = await self._embedder.embed([wanted[i][1] for i in missing])
|
|
193
|
+
if len(fresh) != len(missing):
|
|
194
|
+
raise ValueError(
|
|
195
|
+
f"{self._embedder.embedder_id} returned {len(fresh)} vectors "
|
|
196
|
+
f"for {len(missing)} texts"
|
|
197
|
+
)
|
|
198
|
+
for i, vector in zip(missing, fresh, strict=True):
|
|
199
|
+
self._cache.set(wanted[i][0], vector)
|
|
200
|
+
cached[i] = vector
|
|
201
|
+
_logger.debug(
|
|
202
|
+
"Embedded %d of %d texts; the rest were cached", len(missing), len(wanted)
|
|
203
|
+
)
|
|
204
|
+
return [vector for vector in cached if vector is not None]
|
|
205
|
+
|
|
206
|
+
async def index(self) -> None:
|
|
207
|
+
"""Embed every registered skill, reusing anything already cached."""
|
|
208
|
+
self._corpus = await build_corpus(self._registry)
|
|
209
|
+
self._positive = await self._vectors(
|
|
210
|
+
[(self._key(doc.content_hash, "pos"), doc.positive_text) for doc in self._corpus]
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
negatives = [doc for doc in self._corpus if doc.negative_text]
|
|
214
|
+
vectors = await self._vectors(
|
|
215
|
+
[(self._key(doc.content_hash, "neg"), doc.negative_text) for doc in negatives]
|
|
216
|
+
)
|
|
217
|
+
by_id = dict(zip((doc.skill_id for doc in negatives), vectors, strict=True))
|
|
218
|
+
self._negative = [by_id.get(doc.skill_id) for doc in self._corpus]
|
|
219
|
+
self._indexed_ids = tuple(doc.skill_id for doc in self._corpus)
|
|
220
|
+
|
|
221
|
+
async def _ensure_index(self) -> None:
|
|
222
|
+
if self._indexed_ids != tuple(skill.get_id() for skill in self._registry.list_skills()):
|
|
223
|
+
await self.index()
|
|
224
|
+
|
|
225
|
+
async def select(self, query: str, *, limit: int = DEFAULT_LIMIT) -> Selection:
|
|
226
|
+
"""Return the best *limit* skills for *query*, by cosine similarity."""
|
|
227
|
+
await self._ensure_index()
|
|
228
|
+
|
|
229
|
+
[query_vector] = await self._embedder.embed([query])
|
|
230
|
+
scored = []
|
|
231
|
+
for doc, positive, negative in zip(
|
|
232
|
+
self._corpus, self._positive, self._negative, strict=True
|
|
233
|
+
):
|
|
234
|
+
score = cosine(query_vector, positive)
|
|
235
|
+
if negative is not None:
|
|
236
|
+
score -= self._negative_weight * max(cosine(query_vector, negative), 0.0)
|
|
237
|
+
scored.append(ScoredSkill(doc.skill_id, score))
|
|
238
|
+
scored.sort(key=lambda s: (-s.score, s.skill_id))
|
|
239
|
+
|
|
240
|
+
selected = [s for s in scored if s.score > self._min_score][:limit]
|
|
241
|
+
chosen = {s.skill_id for s in selected}
|
|
242
|
+
selection = Selection(
|
|
243
|
+
query=query,
|
|
244
|
+
selected=selected,
|
|
245
|
+
rejected=[s for s in scored if s.skill_id not in chosen],
|
|
246
|
+
considered=len(scored),
|
|
247
|
+
)
|
|
248
|
+
_logger.info("EmbeddingSelector %s", selection.describe())
|
|
249
|
+
return selection
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""BM25 ranking over skill metadata, with no dependencies at all.
|
|
2
|
+
|
|
3
|
+
This is the default selector, and it is the default because it works
|
|
4
|
+
the moment the package is installed. An embedding selector is better
|
|
5
|
+
at paraphrase; it is also an API key, a network hop and a bill, and a
|
|
6
|
+
package whose only ranker needs all three is a package most people
|
|
7
|
+
never switch on.
|
|
8
|
+
|
|
9
|
+
Scoring is Okapi BM25 over ``description``, ``when_to_use``, the skill
|
|
10
|
+
name and its tags, minus a weighted BM25 over ``when_not_to_use``. A
|
|
11
|
+
skill whose author wrote "not for local test failures" should lose
|
|
12
|
+
ground on a query about a local test failure, not gain it for sharing
|
|
13
|
+
the vocabulary.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import math
|
|
19
|
+
from collections import Counter
|
|
20
|
+
from typing import TYPE_CHECKING
|
|
21
|
+
|
|
22
|
+
from agentskills_core import get_logger
|
|
23
|
+
from agentskills_retrieval.corpus import SkillDocument, build_corpus, tokenize
|
|
24
|
+
from agentskills_retrieval.selector import DEFAULT_LIMIT, ScoredSkill, Selection
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from agentskills_core import SkillRegistry
|
|
28
|
+
|
|
29
|
+
_logger = get_logger(__name__)
|
|
30
|
+
|
|
31
|
+
#: Term-frequency saturation. The standard value; nothing here justifies tuning it.
|
|
32
|
+
K1 = 1.5
|
|
33
|
+
|
|
34
|
+
#: Length normalisation, also the standard value.
|
|
35
|
+
B = 0.75
|
|
36
|
+
|
|
37
|
+
#: How much a ``when_not_to_use`` match counts against a skill.
|
|
38
|
+
#:
|
|
39
|
+
#: Below 1.0 on purpose: a disclaimer is weaker evidence than a
|
|
40
|
+
#: description, because authors write far fewer of them and phrase them
|
|
41
|
+
#: loosely. At 1.0 a single shared word in a disclaimer could cancel a
|
|
42
|
+
#: genuine description match.
|
|
43
|
+
NEGATIVE_WEIGHT = 0.5
|
|
44
|
+
|
|
45
|
+
#: Scores at or below this are treated as no match at all.
|
|
46
|
+
#:
|
|
47
|
+
#: BM25 is unbounded and corpus-relative, so there is no meaningful
|
|
48
|
+
#: absolute floor above zero. Zero still catches the case that matters
|
|
49
|
+
#: — the query shares no term with any skill — but it cannot catch a
|
|
50
|
+
#: query that matches a common word and is nonetheless irrelevant.
|
|
51
|
+
#: That limit is real and is why recall@k is measured rather than assumed.
|
|
52
|
+
DEFAULT_MIN_SCORE = 0.0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class _Bm25Index:
|
|
56
|
+
"""A scored field across the corpus."""
|
|
57
|
+
|
|
58
|
+
def __init__(self, documents: list[list[str]]) -> None:
|
|
59
|
+
self._lengths = [len(terms) for terms in documents]
|
|
60
|
+
self._avg_length = (sum(self._lengths) / len(self._lengths)) if self._lengths else 0.0
|
|
61
|
+
self._frequencies = [Counter(terms) for terms in documents]
|
|
62
|
+
|
|
63
|
+
seen: Counter[str] = Counter()
|
|
64
|
+
for frequency in self._frequencies:
|
|
65
|
+
seen.update(frequency.keys())
|
|
66
|
+
count = len(documents)
|
|
67
|
+
self._idf = {
|
|
68
|
+
# Lucene's variant, which cannot go negative for a term in
|
|
69
|
+
# most documents; the classic form can, and a negative IDF
|
|
70
|
+
# turns a match into a penalty.
|
|
71
|
+
term: math.log(1 + (count - n + 0.5) / (n + 0.5))
|
|
72
|
+
for term, n in seen.items()
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
def score(self, index: int, terms: list[str]) -> float:
|
|
76
|
+
"""Score document *index* against the query *terms*."""
|
|
77
|
+
if not self._avg_length:
|
|
78
|
+
return 0.0
|
|
79
|
+
frequency = self._frequencies[index]
|
|
80
|
+
length = self._lengths[index]
|
|
81
|
+
total = 0.0
|
|
82
|
+
for term in terms:
|
|
83
|
+
occurrences = frequency.get(term, 0)
|
|
84
|
+
if not occurrences:
|
|
85
|
+
continue
|
|
86
|
+
denominator = occurrences + K1 * (1 - B + B * length / self._avg_length)
|
|
87
|
+
total += self._idf[term] * occurrences * (K1 + 1) / denominator
|
|
88
|
+
return total
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class LexicalSelector:
|
|
92
|
+
"""Select skills by BM25 over their catalog metadata.
|
|
93
|
+
|
|
94
|
+
The corpus is built on first use and reused until the registry's
|
|
95
|
+
set of skill IDs changes, so a long-lived agent pays for indexing
|
|
96
|
+
once.
|
|
97
|
+
|
|
98
|
+
Args:
|
|
99
|
+
registry: The registry whose skills are ranked.
|
|
100
|
+
min_score: Scores at or below this are rejected outright. See
|
|
101
|
+
:data:`DEFAULT_MIN_SCORE` for why the default is zero.
|
|
102
|
+
negative_weight: How much a ``when_not_to_use`` match subtracts.
|
|
103
|
+
Pass ``0.0`` to ignore disclaimers entirely.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
def __init__(
|
|
107
|
+
self,
|
|
108
|
+
registry: SkillRegistry,
|
|
109
|
+
*,
|
|
110
|
+
min_score: float = DEFAULT_MIN_SCORE,
|
|
111
|
+
negative_weight: float = NEGATIVE_WEIGHT,
|
|
112
|
+
) -> None:
|
|
113
|
+
self._registry = registry
|
|
114
|
+
self._min_score = min_score
|
|
115
|
+
self._negative_weight = negative_weight
|
|
116
|
+
self._corpus: list[SkillDocument] = []
|
|
117
|
+
self._positive = _Bm25Index([])
|
|
118
|
+
self._negative = _Bm25Index([])
|
|
119
|
+
self._indexed_ids: tuple[str, ...] = ()
|
|
120
|
+
|
|
121
|
+
async def index(self) -> None:
|
|
122
|
+
"""Build the BM25 index, discarding any previous one."""
|
|
123
|
+
self._corpus = await build_corpus(self._registry)
|
|
124
|
+
self._positive = _Bm25Index([doc.positive_terms for doc in self._corpus])
|
|
125
|
+
self._negative = _Bm25Index([doc.negative_terms for doc in self._corpus])
|
|
126
|
+
self._indexed_ids = tuple(doc.skill_id for doc in self._corpus)
|
|
127
|
+
|
|
128
|
+
async def _ensure_index(self) -> None:
|
|
129
|
+
if self._indexed_ids != tuple(skill.get_id() for skill in self._registry.list_skills()):
|
|
130
|
+
await self.index()
|
|
131
|
+
|
|
132
|
+
async def select(self, query: str, *, limit: int = DEFAULT_LIMIT) -> Selection:
|
|
133
|
+
"""Return the best *limit* skills for *query*.
|
|
134
|
+
|
|
135
|
+
Args:
|
|
136
|
+
query: Free text to rank against. The caller decides what
|
|
137
|
+
this is; see the README on why the last user message
|
|
138
|
+
alone is a poor default.
|
|
139
|
+
limit: Maximum number of skills to return.
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
A :class:`~agentskills_retrieval.Selection`, whose
|
|
143
|
+
``skill_ids`` feed ``get_skills_catalog(include=...)``.
|
|
144
|
+
"""
|
|
145
|
+
await self._ensure_index()
|
|
146
|
+
|
|
147
|
+
terms = tokenize(query)
|
|
148
|
+
scored = [
|
|
149
|
+
ScoredSkill(
|
|
150
|
+
doc.skill_id,
|
|
151
|
+
self._positive.score(i, terms)
|
|
152
|
+
- self._negative_weight * self._negative.score(i, terms),
|
|
153
|
+
)
|
|
154
|
+
for i, doc in enumerate(self._corpus)
|
|
155
|
+
]
|
|
156
|
+
# Ties break by ID so the same registry and query always give
|
|
157
|
+
# the same catalog; a prompt that varies run to run is not one
|
|
158
|
+
# you can debug.
|
|
159
|
+
scored.sort(key=lambda s: (-s.score, s.skill_id))
|
|
160
|
+
|
|
161
|
+
selected = [s for s in scored if s.score > self._min_score][:limit]
|
|
162
|
+
chosen = {s.skill_id for s in selected}
|
|
163
|
+
selection = Selection(
|
|
164
|
+
query=query,
|
|
165
|
+
selected=selected,
|
|
166
|
+
rejected=[s for s in scored if s.skill_id not in chosen],
|
|
167
|
+
considered=len(scored),
|
|
168
|
+
)
|
|
169
|
+
_logger.info("LexicalSelector %s", selection.describe())
|
|
170
|
+
return selection
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""What a selector is, and what it returns.
|
|
2
|
+
|
|
3
|
+
A selector answers one question — *given this query, which of these
|
|
4
|
+
skills are worth putting in front of the model?* — and it answers it
|
|
5
|
+
with scores, not a bare list, because a caller that cannot see why a
|
|
6
|
+
skill was dropped cannot debug an agent that failed to use it.
|
|
7
|
+
|
|
8
|
+
The output composes with the filter the catalog already has::
|
|
9
|
+
|
|
10
|
+
selection = await selector.select("the checkout API is down")
|
|
11
|
+
catalog = await registry.get_skills_catalog(
|
|
12
|
+
include=selection.skill_ids,
|
|
13
|
+
total=selection.considered,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
``include=`` is applied before any metadata is fetched, so narrowing
|
|
17
|
+
fifty skills to five also avoids forty-five provider round trips.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from typing import Protocol, runtime_checkable
|
|
24
|
+
|
|
25
|
+
#: Skills a selection returns unless the caller asks for more or fewer.
|
|
26
|
+
DEFAULT_LIMIT = 5
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class ScoredSkill:
|
|
31
|
+
"""One skill and what it scored, in the selector's own units."""
|
|
32
|
+
|
|
33
|
+
skill_id: str
|
|
34
|
+
score: float
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class Selection:
|
|
39
|
+
"""The outcome of one selection, including what it rejected.
|
|
40
|
+
|
|
41
|
+
Attributes:
|
|
42
|
+
query: The text that was ranked against.
|
|
43
|
+
selected: Skills that cleared the floor, best first.
|
|
44
|
+
rejected: Skills that did not, also best first. Kept because
|
|
45
|
+
"the right skill scored just under the floor" and "the
|
|
46
|
+
right skill was not registered" are different bugs and
|
|
47
|
+
look identical without this.
|
|
48
|
+
considered: How many skills were scored, which is the
|
|
49
|
+
denominator to report in the catalog.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
query: str
|
|
53
|
+
selected: list[ScoredSkill]
|
|
54
|
+
rejected: list[ScoredSkill]
|
|
55
|
+
considered: int
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def skill_ids(self) -> list[str]:
|
|
59
|
+
"""The selected IDs, ready to pass to ``include=``."""
|
|
60
|
+
return [scored.skill_id for scored in self.selected]
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def is_empty(self) -> bool:
|
|
64
|
+
"""Whether nothing cleared the floor."""
|
|
65
|
+
return not self.selected
|
|
66
|
+
|
|
67
|
+
def describe(self) -> str:
|
|
68
|
+
"""Return a one-line summary with scores, for logs and reports."""
|
|
69
|
+
if self.is_empty:
|
|
70
|
+
best = f", best rejected {self.rejected[0].score:.3f}" if self.rejected else ""
|
|
71
|
+
return f"no skill cleared the floor for {self.query!r} of {self.considered}{best}"
|
|
72
|
+
scores = ", ".join(f"{s.skill_id}={s.score:.3f}" for s in self.selected)
|
|
73
|
+
return f"selected {len(self.selected)} of {self.considered} for {self.query!r}: {scores}"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@runtime_checkable
|
|
77
|
+
class SkillSelector(Protocol):
|
|
78
|
+
"""Rank registered skills against a query.
|
|
79
|
+
|
|
80
|
+
One method, so a caller can supply its own ranker — a hosted
|
|
81
|
+
reranking API, a hand-written routing table — without inheriting
|
|
82
|
+
anything from this package.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
async def select(self, query: str, *, limit: int = DEFAULT_LIMIT) -> Selection:
|
|
86
|
+
"""Return the best *limit* skills for *query*, with scores."""
|
|
87
|
+
...
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "agentskills-retrieval"
|
|
3
|
+
version = "0.5.0"
|
|
4
|
+
description = "Query-time skill selection for the Agent Skills format (https://agentskills.io)"
|
|
5
|
+
license = "MIT"
|
|
6
|
+
authors = ["Pratik Panda"]
|
|
7
|
+
readme = "README.md"
|
|
8
|
+
homepage = "https://agentskills.io"
|
|
9
|
+
repository = "https://github.com/pratikxpanda/agentskills-sdk"
|
|
10
|
+
packages = [{include = "agentskills_retrieval"}]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 3 - Alpha",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"Topic :: Software Development :: Libraries",
|
|
15
|
+
"Programming Language :: Python :: 3.12",
|
|
16
|
+
"Programming Language :: Python :: 3.13",
|
|
17
|
+
"Programming Language :: Python :: 3.14",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
[tool.poetry.dependencies]
|
|
21
|
+
python = ">=3.12,<4.0"
|
|
22
|
+
agentskills-core = ">=0.5.0,<1.0"
|
|
23
|
+
|
|
24
|
+
[build-system]
|
|
25
|
+
requires = ["poetry-core"]
|
|
26
|
+
build-backend = "poetry.core.masonry.api"
|