willuri 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- willuri-0.1.0/LICENSE +110 -0
- willuri-0.1.0/PKG-INFO +57 -0
- willuri-0.1.0/README.md +45 -0
- willuri-0.1.0/pyproject.toml +31 -0
- willuri-0.1.0/setup.cfg +4 -0
- willuri-0.1.0/tests/test_dsl_and_models.py +57 -0
- willuri-0.1.0/tests/test_router.py +125 -0
- willuri-0.1.0/tests/test_ssot_config.py +50 -0
- willuri-0.1.0/willuri/__init__.py +29 -0
- willuri-0.1.0/willuri/cli.py +58 -0
- willuri-0.1.0/willuri/client.py +37 -0
- willuri-0.1.0/willuri/domains.py +93 -0
- willuri-0.1.0/willuri/dsl.py +53 -0
- willuri-0.1.0/willuri/models.py +90 -0
- willuri-0.1.0/willuri/prompts.py +62 -0
- willuri-0.1.0/willuri/router.py +155 -0
- willuri-0.1.0/willuri/semantic.py +47 -0
- willuri-0.1.0/willuri.egg-info/PKG-INFO +57 -0
- willuri-0.1.0/willuri.egg-info/SOURCES.txt +21 -0
- willuri-0.1.0/willuri.egg-info/dependency_links.txt +1 -0
- willuri-0.1.0/willuri.egg-info/entry_points.txt +2 -0
- willuri-0.1.0/willuri.egg-info/requires.txt +1 -0
- willuri-0.1.0/willuri.egg-info/top_level.txt +1 -0
willuri-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
PROPRIETARY AND CONFIDENTIAL SOURCE CODE LICENSE
|
|
2
|
+
STRICT ANTI-AI / ANTI-MINING RESERVATION OF RIGHTS
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2026 Tomasz Sapletta Prototypowanie.pl NIP: 5881918662, REGON: 220665410. All rights reserved.
|
|
5
|
+
Właścicielem oprogramowania jest Tomasz Sapletta Prototypowanie.pl NIP: 5881918662, REGON: 220665410.
|
|
6
|
+
|
|
7
|
+
================================================================================
|
|
8
|
+
IMPORTANT NOTICE - PROPRIETARY AND CONFIDENTIAL
|
|
9
|
+
================================================================================
|
|
10
|
+
|
|
11
|
+
This software, source code, documentation, specifications, schemas, models,
|
|
12
|
+
algorithms, and associated digital assets ("the Software") are the confidential
|
|
13
|
+
and proprietary intellectual property of Tomasz Sapletta Prototypowanie.pl
|
|
14
|
+
(NIP: 5881918662, REGON: 220665410) and its licensors.
|
|
15
|
+
|
|
16
|
+
The Software is NOT licensed under any open source, free software, or
|
|
17
|
+
source-available license (such as MIT, Apache 2.0, BSD, GPL, AGPL, BSL, or SSPL).
|
|
18
|
+
All rights not expressly granted under a separate, signed written commercial
|
|
19
|
+
agreement executed by Tomasz Sapletta Prototypowanie.pl are strictly reserved.
|
|
20
|
+
|
|
21
|
+
================================================================================
|
|
22
|
+
SECTION 1. STRICT PROHIBITION ON ARTIFICIAL INTELLIGENCE & MACHINE LEARNING INGESTION
|
|
23
|
+
================================================================================
|
|
24
|
+
|
|
25
|
+
Under no circumstances may any portion of the Software, whether in human-readable
|
|
26
|
+
source code, compiled binary, abstract syntax tree (AST), intermediate representation,
|
|
27
|
+
documentation, ticket history, or commit metadata, be:
|
|
28
|
+
|
|
29
|
+
1.1. Ingested, scraped, crawled, harvested, indexed, tokenized, cached, or stored
|
|
30
|
+
for the purpose of training, pre-training, fine-tuning, evaluating, or weighting
|
|
31
|
+
any artificial intelligence system, machine learning model, large language
|
|
32
|
+
model (LLM), code-generation model, neural network, diffusion model, or
|
|
33
|
+
multimodal model.
|
|
34
|
+
|
|
35
|
+
1.2. Used by, or fed into the training datasets, prompt contexts, retrieval-augmented
|
|
36
|
+
generation (RAG) stores, vector databases, or telemetry caches of any commercial
|
|
37
|
+
or non-commercial AI system, including but not limited to:
|
|
38
|
+
- GitHub Copilot / Microsoft Copilot (Microsoft Corporation)
|
|
39
|
+
- OpenAI ChatGPT, Codex, GPT-4, GPT-5, o-series, and related APIs (OpenAI Inc.)
|
|
40
|
+
- Google Gemini, PaLM, Codey, Vertex AI, and DeepMind systems (Google LLC / Alphabet)
|
|
41
|
+
- Anthropic Claude, Constitutional AI, and related models (Anthropic PBC)
|
|
42
|
+
- Meta LLaMA, Code Llama, and related foundation models (Meta Platforms, Inc.)
|
|
43
|
+
- Amazon CodeWhisperer, Bedrock, and Titan models (Amazon.com, Inc.)
|
|
44
|
+
- Mistral AI, Cohere, Cursor AI, Tabnine, Replit Ghostwriter, or any automated
|
|
45
|
+
code-synthesis engine or AI coding assistant.
|
|
46
|
+
|
|
47
|
+
1.3. Processed or analyzed by any automated tool, bot, crawler, or scraper to generate
|
|
48
|
+
synthetic code, code suggestions, autocomplete suggestions, embeddings, or
|
|
49
|
+
derivative software models.
|
|
50
|
+
|
|
51
|
+
================================================================================
|
|
52
|
+
SECTION 2. STATUTORY RESERVATION OF RIGHTS (EU DIRECTIVE 2019/790)
|
|
53
|
+
================================================================================
|
|
54
|
+
|
|
55
|
+
2.1. In accordance with Article 4(3) of Directive (EU) 2019/790 of the European
|
|
56
|
+
Parliament and of the Council on copyright and related rights in the Digital
|
|
57
|
+
Single Market, the rights holder hereby EXPRESSLY RESERVES ALL RIGHTS
|
|
58
|
+
regarding Text and Data Mining (TDM) for this Software, its source repositories,
|
|
59
|
+
documentation, schemas, and all associated materials.
|
|
60
|
+
|
|
61
|
+
2.2. Any extraction, automated reading, indexing, or reproduction of the Software
|
|
62
|
+
for text and data mining purposes within the meaning of Directive (EU) 2019/790
|
|
63
|
+
is strictly prohibited without prior written authorization from the rights holder.
|
|
64
|
+
|
|
65
|
+
2.3. This reservation applies worldwide and is declared in machine-readable form
|
|
66
|
+
within the repository metadata, headers, and this License specification.
|
|
67
|
+
|
|
68
|
+
================================================================================
|
|
69
|
+
SECTION 3. HOSTING & PLATFORM PREEMPTION CLAUSE
|
|
70
|
+
================================================================================
|
|
71
|
+
|
|
72
|
+
3.1. The presence of this source code or repository on any third-party hosting service,
|
|
73
|
+
code forge, version control platform, or cloud provider (including but not
|
|
74
|
+
limited to GitHub, GitLab, Bitbucket, or AWS) does NOT constitute a waiver,
|
|
75
|
+
license grant, or implied consent to the hosting provider, its parent,
|
|
76
|
+
subsidiaries, or third parties to:
|
|
77
|
+
(a) use the Software for machine learning training, telemetry harvesting, or
|
|
78
|
+
model improvement;
|
|
79
|
+
(b) distribute or reproduce the Software to other users or automated agents;
|
|
80
|
+
(c) index the Software in any public search or retrieval system.
|
|
81
|
+
|
|
82
|
+
3.2. Any terms of service, platform agreements, or automated licenses purporting
|
|
83
|
+
to grant the platform operator an implied, royalty-free, or transferable
|
|
84
|
+
license to utilize hosted code for artificial intelligence development, model
|
|
85
|
+
training, or feature synthesis are hereby explicitly rejected, preempted,
|
|
86
|
+
and declared null and void with respect to this Software.
|
|
87
|
+
|
|
88
|
+
================================================================================
|
|
89
|
+
SECTION 4. PROHIBITED USES & RESTRICTIONS
|
|
90
|
+
================================================================================
|
|
91
|
+
|
|
92
|
+
Except as expressly authorized under a valid, executed commercial enterprise license
|
|
93
|
+
with Tomasz Sapletta Prototypowanie.pl, you may NOT:
|
|
94
|
+
- Copy, modify, adapt, translate, reverse engineer, decompile, or disassemble the Software.
|
|
95
|
+
- Distribute, sublicense, rent, lease, lend, sell, or publicly display the Software.
|
|
96
|
+
- Use the Software to provide competitive time-sharing, SaaS, cloud twin, or virtualization services.
|
|
97
|
+
- Remove, alter, or obscure any proprietary notices, copyright labels, or license files.
|
|
98
|
+
|
|
99
|
+
================================================================================
|
|
100
|
+
SECTION 5. ENFORCEMENT & REMEDIES
|
|
101
|
+
================================================================================
|
|
102
|
+
|
|
103
|
+
Any breach of this License, including unauthorized scraping, tokenization, or ingestion
|
|
104
|
+
into an artificial intelligence system, constitutes willful copyright infringement,
|
|
105
|
+
unauthorized access, and breach of contract. Tomasz Sapletta Prototypowanie.pl reserves
|
|
106
|
+
the right to seek immediate injunctive relief, statutory damages, compensation for commercial
|
|
107
|
+
harm, and mandatory destruction of all derivative model weights, embeddings, and training
|
|
108
|
+
caches containing representations of the Software.
|
|
109
|
+
|
|
110
|
+
For licensing inquiries and commercial authorization, contact: tom@prototypowanie.pl / legal@clonerd.com
|
willuri-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: willuri
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Natural language to specialized URI process router with domain-specific small LLMs
|
|
5
|
+
Author-email: Tom Sapletta <tom@sapletta.com>
|
|
6
|
+
License: Proprietary
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: pyyaml>=6.0
|
|
11
|
+
Dynamic: license-file
|
|
12
|
+
|
|
13
|
+
# willuri
|
|
14
|
+
|
|
15
|
+
Natural language to specialized process URI router powered by domain-specific lightweight LLMs.
|
|
16
|
+
|
|
17
|
+
## Overview
|
|
18
|
+
|
|
19
|
+
In the Paxlet / Willman ecosystem, every capability is addressed by an explicit process URI (`willman://operation/...`, `dockuri://...`).
|
|
20
|
+
`willuri` maps natural language user instructions or tickets from the board to specialized process URIs using small, fast, domain-targeted LLMs instead of a monolithic general-purpose model.
|
|
21
|
+
|
|
22
|
+
### Specialized Domain Models
|
|
23
|
+
|
|
24
|
+
| Domain | Process URI Scope | Primary Model | Alternative | Focus |
|
|
25
|
+
| :--- | :--- | :--- | :--- | :--- |
|
|
26
|
+
| **`code`** | `willman://operation/code.*`, `git.resolve_conflicts` | `qwen2.5-coder:3b` | `willman-nlp:qwen2.5-3b` | Code synthesis, AST edits, diffs, bug repairs |
|
|
27
|
+
| **`api_ops`** | `willman://operation/github.*`, `dockuri.*`, `files.*` | `granite4.1:3b` | `willman-nlp:qwen2.5-3b` | Structured API tools, JSON schema, enterprise ops |
|
|
28
|
+
| **`planning`** | `willman://operation/koru.*`, `planfile.*`, `schedule.*` | `llama3.2:3b` | `willman-nlp:qwen2.5-3b` | Ticket decomposition, sprint planning, DAG subtasks |
|
|
29
|
+
| **`text`** | `willman://operation/text.*` | `willman-nlp:qwen2.5-1.5b` | `willman-nlp:qwen2.5-3b` | String normalization, fast formatting, prose |
|
|
30
|
+
|
|
31
|
+
## Usage
|
|
32
|
+
|
|
33
|
+
### Python API
|
|
34
|
+
```python
|
|
35
|
+
from willuri.router import UriRouter
|
|
36
|
+
|
|
37
|
+
router = UriRouter()
|
|
38
|
+
router.add_operation("willman://operation/text.normalize/v1", "Normalize whitespace")
|
|
39
|
+
router.add_operation("willman://operation/git.resolve_conflicts/v1", "Resolve git merge conflicts")
|
|
40
|
+
|
|
41
|
+
result = router.route("Rozwiąż konflikty git w repozytorium")
|
|
42
|
+
print(result["uri"])
|
|
43
|
+
# => "willman://operation/git.resolve_conflicts/v1"
|
|
44
|
+
print(result["model_used"])
|
|
45
|
+
# => "qwen2.5-coder:3b"
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### CLI
|
|
49
|
+
```bash
|
|
50
|
+
willuri classify "Popraw błąd w funkcji liczącej sumę"
|
|
51
|
+
willuri route "Znormalizuj odstępy w tekście 'Ala ma kota'"
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Verification
|
|
55
|
+
```bash
|
|
56
|
+
python3 -m unittest discover -s tests -v
|
|
57
|
+
```
|
willuri-0.1.0/README.md
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# willuri
|
|
2
|
+
|
|
3
|
+
Natural language to specialized process URI router powered by domain-specific lightweight LLMs.
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
In the Paxlet / Willman ecosystem, every capability is addressed by an explicit process URI (`willman://operation/...`, `dockuri://...`).
|
|
8
|
+
`willuri` maps natural language user instructions or tickets from the board to specialized process URIs using small, fast, domain-targeted LLMs instead of a monolithic general-purpose model.
|
|
9
|
+
|
|
10
|
+
### Specialized Domain Models
|
|
11
|
+
|
|
12
|
+
| Domain | Process URI Scope | Primary Model | Alternative | Focus |
|
|
13
|
+
| :--- | :--- | :--- | :--- | :--- |
|
|
14
|
+
| **`code`** | `willman://operation/code.*`, `git.resolve_conflicts` | `qwen2.5-coder:3b` | `willman-nlp:qwen2.5-3b` | Code synthesis, AST edits, diffs, bug repairs |
|
|
15
|
+
| **`api_ops`** | `willman://operation/github.*`, `dockuri.*`, `files.*` | `granite4.1:3b` | `willman-nlp:qwen2.5-3b` | Structured API tools, JSON schema, enterprise ops |
|
|
16
|
+
| **`planning`** | `willman://operation/koru.*`, `planfile.*`, `schedule.*` | `llama3.2:3b` | `willman-nlp:qwen2.5-3b` | Ticket decomposition, sprint planning, DAG subtasks |
|
|
17
|
+
| **`text`** | `willman://operation/text.*` | `willman-nlp:qwen2.5-1.5b` | `willman-nlp:qwen2.5-3b` | String normalization, fast formatting, prose |
|
|
18
|
+
|
|
19
|
+
## Usage
|
|
20
|
+
|
|
21
|
+
### Python API
|
|
22
|
+
```python
|
|
23
|
+
from willuri.router import UriRouter
|
|
24
|
+
|
|
25
|
+
router = UriRouter()
|
|
26
|
+
router.add_operation("willman://operation/text.normalize/v1", "Normalize whitespace")
|
|
27
|
+
router.add_operation("willman://operation/git.resolve_conflicts/v1", "Resolve git merge conflicts")
|
|
28
|
+
|
|
29
|
+
result = router.route("Rozwiąż konflikty git w repozytorium")
|
|
30
|
+
print(result["uri"])
|
|
31
|
+
# => "willman://operation/git.resolve_conflicts/v1"
|
|
32
|
+
print(result["model_used"])
|
|
33
|
+
# => "qwen2.5-coder:3b"
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
### CLI
|
|
37
|
+
```bash
|
|
38
|
+
willuri classify "Popraw błąd w funkcji liczącej sumę"
|
|
39
|
+
willuri route "Znormalizuj odstępy w tekście 'Ala ma kota'"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Verification
|
|
43
|
+
```bash
|
|
44
|
+
python3 -m unittest discover -s tests -v
|
|
45
|
+
```
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "willuri"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Natural language to specialized URI process router with domain-specific small LLMs"
|
|
9
|
+
authors = [{ name = "Tom Sapletta", email = "tom@sapletta.com" }]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
license = { text = "Proprietary" }
|
|
13
|
+
dependencies = [
|
|
14
|
+
"pyyaml>=6.0",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.scripts]
|
|
18
|
+
willuri = "willuri.cli:main"
|
|
19
|
+
|
|
20
|
+
[tool.setuptools.packages.find]
|
|
21
|
+
where = ["."]
|
|
22
|
+
include = ["willuri*"]
|
|
23
|
+
|
|
24
|
+
[tool.pytest.ini_options]
|
|
25
|
+
pythonpath = ["."]
|
|
26
|
+
addopts = "-p wellmanifest_governance"
|
|
27
|
+
|
|
28
|
+
[tool.wellmanifest]
|
|
29
|
+
standard = "0.20.81"
|
|
30
|
+
revision = "525a0d285c9054f079f0aa22f5c41931ce90afb8"
|
|
31
|
+
gate = "project/governance-check.sh"
|
willuri-0.1.0/setup.cfg
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Unit tests for willuri DSL, models, and semantic matching."""
|
|
2
|
+
import pytest
|
|
3
|
+
from willuri import (
|
|
4
|
+
parse_uri,
|
|
5
|
+
with_params,
|
|
6
|
+
format_uri,
|
|
7
|
+
validate_uri,
|
|
8
|
+
MODEL_SPECS,
|
|
9
|
+
get_model_for_domain,
|
|
10
|
+
generate_modelfile,
|
|
11
|
+
SemanticMatcher,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_dsl_with_params_and_parse():
|
|
16
|
+
base = "willman://operation/git.status"
|
|
17
|
+
params = {"path": "~/github/willman", "short": True}
|
|
18
|
+
uri = with_params(base, params)
|
|
19
|
+
assert uri.startswith("willman://operation/git.status?")
|
|
20
|
+
assert "path" in uri
|
|
21
|
+
|
|
22
|
+
parsed_base, parsed_params = parse_uri(uri)
|
|
23
|
+
assert parsed_base == base
|
|
24
|
+
assert parsed_params["path"] == "~/github/willman"
|
|
25
|
+
assert parsed_params["short"] is True
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_dsl_format_and_validate():
|
|
29
|
+
uri = format_uri("dockuri", "container", "redis", {"port": 6379})
|
|
30
|
+
assert uri == "dockuri://container/redis?port=6379"
|
|
31
|
+
assert validate_uri(uri) is True
|
|
32
|
+
assert validate_uri("invalid://something#frag") is False
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_model_specs():
|
|
36
|
+
coder = MODEL_SPECS["qwen2.5-coder:3b"]
|
|
37
|
+
assert coder.parameters == "3B"
|
|
38
|
+
assert coder.context_tokens == 8192
|
|
39
|
+
assert "code" in coder.domains
|
|
40
|
+
|
|
41
|
+
model = get_model_for_domain("code")
|
|
42
|
+
assert model == "qwen2.5-coder:3b"
|
|
43
|
+
|
|
44
|
+
modelfile = generate_modelfile(model)
|
|
45
|
+
assert "FROM qwen2.5-coder:3b" in modelfile
|
|
46
|
+
assert "num_ctx 8192" in modelfile
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_semantic_matcher():
|
|
50
|
+
catalog = [
|
|
51
|
+
{"uri": "willman://operation/git.status", "description": "Sprawdź status repozytorium git i zmodyfikowane pliki"},
|
|
52
|
+
{"uri": "willman://operation/dockuri.run", "description": "Uruchom kontener dockuri w środowisku sandbox"},
|
|
53
|
+
]
|
|
54
|
+
matcher = SemanticMatcher(catalog)
|
|
55
|
+
results = matcher.match("jaki jest git status plikow?")
|
|
56
|
+
assert len(results) > 0
|
|
57
|
+
assert results[0][0]["uri"] == "willman://operation/git.status"
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import unittest
|
|
2
|
+
from willuri.domains import classify_domain, DOMAIN_REGISTRY
|
|
3
|
+
from willuri.router import UriRouter
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class UriRouterTest(unittest.TestCase):
|
|
7
|
+
def test_domain_classification_code(self):
|
|
8
|
+
domain = classify_domain("Popraw błąd w funkcji liczącej sumę")
|
|
9
|
+
self.assertEqual(domain.name, "code")
|
|
10
|
+
self.assertIn("qwen2.5-coder:3b", domain.preferred_models)
|
|
11
|
+
|
|
12
|
+
def test_domain_classification_api(self):
|
|
13
|
+
domain = classify_domain("Wyszukaj otwarte zgłoszenie w repozytorium na githubie")
|
|
14
|
+
self.assertEqual(domain.name, "api_ops")
|
|
15
|
+
self.assertIn("granite4.1:3b", domain.preferred_models)
|
|
16
|
+
|
|
17
|
+
def test_domain_classification_planning(self):
|
|
18
|
+
domain = classify_domain("Rozdziel ticket i dodaj podzadania w sprincie")
|
|
19
|
+
self.assertEqual(domain.name, "planning")
|
|
20
|
+
self.assertIn("llama3.2:3b", domain.preferred_models)
|
|
21
|
+
|
|
22
|
+
def test_domain_classification_text(self):
|
|
23
|
+
domain = classify_domain("Usuń nadmiarowe odstępy w tekście")
|
|
24
|
+
self.assertEqual(domain.name, "text")
|
|
25
|
+
self.assertIn("willman-nlp:qwen2.5-3b", domain.preferred_models)
|
|
26
|
+
|
|
27
|
+
def test_domain_classification_email(self):
|
|
28
|
+
domain = classify_domain("Wyszukaj profile i pobierz e-maile z aplikacji Thunderbird")
|
|
29
|
+
self.assertEqual(domain.name, "api_ops")
|
|
30
|
+
self.assertIn("granite4.1:3b", domain.preferred_models)
|
|
31
|
+
|
|
32
|
+
def test_recommend_model(self):
|
|
33
|
+
from willuri.domains import recommend_model
|
|
34
|
+
model_code = recommend_model("Napisz skrypt w pythonie liczący silnię")
|
|
35
|
+
self.assertEqual(model_code, "qwen2.5-coder:3b")
|
|
36
|
+
model_text = recommend_model("Znormalizuj tekst i usuń spacje")
|
|
37
|
+
self.assertEqual(model_text, "willman-nlp:qwen2.5-3b")
|
|
38
|
+
|
|
39
|
+
def test_router_adds_operation(self):
|
|
40
|
+
router = UriRouter()
|
|
41
|
+
router.add_operation("willman://operation/text.normalize/v1", "Normalize whitespace")
|
|
42
|
+
self.assertEqual(len(router.catalog), 1)
|
|
43
|
+
self.assertEqual(router.catalog[0]["uri"], "willman://operation/text.normalize/v1")
|
|
44
|
+
|
|
45
|
+
def test_empty_catalog_unsupported(self):
|
|
46
|
+
router = UriRouter(catalog=[])
|
|
47
|
+
res = router.route("Normalize some text")
|
|
48
|
+
self.assertEqual(res["status"], "unsupported")
|
|
49
|
+
self.assertEqual(res["error"], "empty_catalog")
|
|
50
|
+
|
|
51
|
+
def test_reject_invented_unregistered_uri(self):
|
|
52
|
+
# Provider invents a URI not present in catalog
|
|
53
|
+
def mock_provider(**kw):
|
|
54
|
+
return {"uri": "willman://operation/invented.fake/v1", "input": {}}
|
|
55
|
+
|
|
56
|
+
router = UriRouter(
|
|
57
|
+
catalog=[{"uri": "willman://operation/text.normalize/v1", "description": "Normalize whitespace"}],
|
|
58
|
+
provider=mock_provider,
|
|
59
|
+
)
|
|
60
|
+
res = router.route("Normalize text")
|
|
61
|
+
self.assertEqual(res["status"], "unsupported")
|
|
62
|
+
self.assertEqual(res["error"], "unregistered_uri")
|
|
63
|
+
|
|
64
|
+
def test_schema_valid_arguments_yields_ok(self):
|
|
65
|
+
schema = {
|
|
66
|
+
"type": "object",
|
|
67
|
+
"required": ["path"],
|
|
68
|
+
"properties": {
|
|
69
|
+
"path": {"type": "string"},
|
|
70
|
+
"short": {"type": "boolean"},
|
|
71
|
+
},
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
def mock_provider(**kw):
|
|
75
|
+
return {
|
|
76
|
+
"uri": "willman://operation/git.status",
|
|
77
|
+
"input": {"path": "/tmp/repo", "short": True},
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
router = UriRouter(
|
|
81
|
+
catalog=[{
|
|
82
|
+
"uri": "willman://operation/git.status",
|
|
83
|
+
"description": "Git status check",
|
|
84
|
+
"input_schema": schema,
|
|
85
|
+
}],
|
|
86
|
+
provider=mock_provider,
|
|
87
|
+
)
|
|
88
|
+
res = router.route("git status /tmp/repo")
|
|
89
|
+
self.assertEqual(res["status"], "ok")
|
|
90
|
+
self.assertEqual(res["uri"], "willman://operation/git.status")
|
|
91
|
+
self.assertEqual(res["input"]["path"], "/tmp/repo")
|
|
92
|
+
|
|
93
|
+
def test_schema_invalid_arguments_yields_clarify(self):
|
|
94
|
+
schema = {
|
|
95
|
+
"type": "object",
|
|
96
|
+
"required": ["path"],
|
|
97
|
+
"properties": {
|
|
98
|
+
"path": {"type": "string"},
|
|
99
|
+
},
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
# Missing required argument 'path'
|
|
103
|
+
def mock_provider(**kw):
|
|
104
|
+
return {
|
|
105
|
+
"uri": "willman://operation/git.status",
|
|
106
|
+
"input": {"wrong_field": 123},
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
router = UriRouter(
|
|
110
|
+
catalog=[{
|
|
111
|
+
"uri": "willman://operation/git.status",
|
|
112
|
+
"description": "Git status check",
|
|
113
|
+
"input_schema": schema,
|
|
114
|
+
}],
|
|
115
|
+
provider=mock_provider,
|
|
116
|
+
)
|
|
117
|
+
res = router.route("git status")
|
|
118
|
+
self.assertEqual(res["status"], "clarify")
|
|
119
|
+
self.assertEqual(res["error"], "invalid_arguments")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
if __name__ == "__main__":
|
|
123
|
+
unittest.main()
|
|
124
|
+
|
|
125
|
+
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Regression test suite enforcing Single Source of Truth and no hardcoded deployment values.
|
|
2
|
+
|
|
3
|
+
Conforms to wellmanifest/ssot standard (SSOT-CONFIG-001):
|
|
4
|
+
- No private network IP literals (e.g. 192.168.x.x) in willuri/ and tests/.
|
|
5
|
+
- No hardcoded /home/ user paths in source files or tests.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import ipaddress
|
|
10
|
+
import re
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
import unittest
|
|
13
|
+
|
|
14
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class SsotConfigRegressionTests(unittest.TestCase):
|
|
18
|
+
def test_no_private_network_literals_in_source(self):
|
|
19
|
+
"""Source and test files must not hardcode LAN or private IP addresses."""
|
|
20
|
+
source_dirs = [ROOT / "willuri", ROOT / "tests"]
|
|
21
|
+
py_files = [p for d in source_dirs for p in d.rglob("*.py")]
|
|
22
|
+
|
|
23
|
+
hits = []
|
|
24
|
+
for f in py_files:
|
|
25
|
+
text = f.read_text(encoding="utf-8")
|
|
26
|
+
for m in re.finditer(r"\b(\d{1,3}(?:\.\d{1,3}){3})\b", text):
|
|
27
|
+
ip_str = m.group(1)
|
|
28
|
+
try:
|
|
29
|
+
ip = ipaddress.ip_address(ip_str)
|
|
30
|
+
if ip.is_private and not ip.is_loopback and not ip.is_unspecified:
|
|
31
|
+
hits.append(f"{f.relative_to(ROOT)}: {ip_str}")
|
|
32
|
+
except ValueError:
|
|
33
|
+
pass
|
|
34
|
+
self.assertEqual(hits, [], "Private network IPs hardcoded in source:\n" + "\n".join(hits))
|
|
35
|
+
|
|
36
|
+
def test_no_hardcoded_home_paths_in_source(self):
|
|
37
|
+
"""Source and test files must not hardcode /home/<user> paths."""
|
|
38
|
+
source_dirs = [ROOT / "willuri", ROOT / "tests"]
|
|
39
|
+
py_files = [p for d in source_dirs for p in d.rglob("*.py")]
|
|
40
|
+
|
|
41
|
+
hits = []
|
|
42
|
+
for f in py_files:
|
|
43
|
+
text = f.read_text(encoding="utf-8")
|
|
44
|
+
for m in re.finditer(r"""(?:"|')/home/[a-zA-Z0-9_-]+(?:/|["'])""", text):
|
|
45
|
+
hits.append(f"{f.relative_to(ROOT)}: {m.group(0)}")
|
|
46
|
+
self.assertEqual(hits, [], "Hardcoded /home/<user> paths in source:\n" + "\n".join(hits))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
if __name__ == "__main__":
|
|
50
|
+
unittest.main()
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""willuri: Universal NL, URI, and specialized LLM router SSOT for Paxlet / Willman."""
|
|
2
|
+
from .domains import DOMAIN_REGISTRY, DomainProfile, classify_domain
|
|
3
|
+
from .router import UriRouter
|
|
4
|
+
from .client import query_ollama
|
|
5
|
+
from .dsl import parse_uri, with_params, format_uri, validate_uri
|
|
6
|
+
from .models import MODEL_SPECS, get_model_for_domain, generate_modelfile
|
|
7
|
+
from .prompts import build_routing_prompt, build_decomposition_prompt
|
|
8
|
+
from .semantic import SemanticMatcher, compute_similarity, tokenize
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
|
11
|
+
__all__ = [
|
|
12
|
+
"UriRouter",
|
|
13
|
+
"classify_domain",
|
|
14
|
+
"DOMAIN_REGISTRY",
|
|
15
|
+
"DomainProfile",
|
|
16
|
+
"query_ollama",
|
|
17
|
+
"parse_uri",
|
|
18
|
+
"with_params",
|
|
19
|
+
"format_uri",
|
|
20
|
+
"validate_uri",
|
|
21
|
+
"MODEL_SPECS",
|
|
22
|
+
"get_model_for_domain",
|
|
23
|
+
"generate_modelfile",
|
|
24
|
+
"build_routing_prompt",
|
|
25
|
+
"build_decomposition_prompt",
|
|
26
|
+
"SemanticMatcher",
|
|
27
|
+
"compute_similarity",
|
|
28
|
+
"tokenize",
|
|
29
|
+
]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""CLI for willuri router."""
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import sys
|
|
5
|
+
from .router import UriRouter
|
|
6
|
+
from .domains import classify_domain
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main():
|
|
10
|
+
parser = argparse.ArgumentParser(prog="willuri", description="Willuri: NL to URI Process Router")
|
|
11
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
12
|
+
|
|
13
|
+
sub.add_parser("domains", help="List all registered domains and specialized models")
|
|
14
|
+
|
|
15
|
+
classify_cmd = sub.add_parser("classify", help="Classify instruction domain")
|
|
16
|
+
classify_cmd.add_argument("instruction", help="Natural language instruction")
|
|
17
|
+
|
|
18
|
+
route_cmd = sub.add_parser("route", help="Route instruction to URI via specialized LLM")
|
|
19
|
+
route_cmd.add_argument("instruction", help="Natural language instruction")
|
|
20
|
+
route_cmd.add_argument("--model", help="Override LLM model")
|
|
21
|
+
route_cmd.add_argument("--input", default="{}", help="Input JSON data")
|
|
22
|
+
|
|
23
|
+
args = parser.parse_args()
|
|
24
|
+
|
|
25
|
+
if args.command == "domains":
|
|
26
|
+
from .domains import DOMAIN_REGISTRY
|
|
27
|
+
res = {
|
|
28
|
+
name: {
|
|
29
|
+
"description": prof.description,
|
|
30
|
+
"preferred_models": prof.preferred_models,
|
|
31
|
+
"uri_prefixes": prof.uri_prefixes,
|
|
32
|
+
"keywords": prof.keywords,
|
|
33
|
+
}
|
|
34
|
+
for name, prof in DOMAIN_REGISTRY.items()
|
|
35
|
+
}
|
|
36
|
+
print(json.dumps(res, indent=2, ensure_ascii=False))
|
|
37
|
+
return 0
|
|
38
|
+
|
|
39
|
+
if args.command == "classify":
|
|
40
|
+
domain = classify_domain(args.instruction)
|
|
41
|
+
print(json.dumps({
|
|
42
|
+
"domain": domain.name,
|
|
43
|
+
"description": domain.description,
|
|
44
|
+
"preferred_models": domain.preferred_models,
|
|
45
|
+
"uri_prefixes": domain.uri_prefixes,
|
|
46
|
+
}, indent=2, ensure_ascii=False))
|
|
47
|
+
return 0
|
|
48
|
+
|
|
49
|
+
if args.command == "route":
|
|
50
|
+
input_data = json.loads(args.input)
|
|
51
|
+
router = UriRouter()
|
|
52
|
+
result = router.route(args.instruction, input_data=input_data, model_override=args.model)
|
|
53
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
54
|
+
return 0 if result.get("status") in ("ok", "routed") else 1
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
if __name__ == "__main__":
|
|
58
|
+
sys.exit(main())
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Specialized LLM inference client for URI parameter extraction and routing."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import urllib.request
|
|
6
|
+
from typing import Any, Dict, List, Optional
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def query_ollama(
|
|
10
|
+
model: str,
|
|
11
|
+
prompt: str,
|
|
12
|
+
schema: Optional[Dict[str, Any]] = None,
|
|
13
|
+
timeout: float = 30.0,
|
|
14
|
+
host: str = "http://127.0.0.1:11434",
|
|
15
|
+
) -> Dict[str, Any]:
|
|
16
|
+
"""Execute a structured generation query against local Ollama endpoint."""
|
|
17
|
+
url = f"{host.rstrip('/')}/api/generate"
|
|
18
|
+
payload: Dict[str, Any] = {
|
|
19
|
+
"model": model,
|
|
20
|
+
"prompt": prompt,
|
|
21
|
+
"stream": False,
|
|
22
|
+
}
|
|
23
|
+
if schema:
|
|
24
|
+
payload["format"] = schema
|
|
25
|
+
else:
|
|
26
|
+
payload["format"] = "json"
|
|
27
|
+
|
|
28
|
+
data = json.dumps(payload).encode("utf-8")
|
|
29
|
+
req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json"})
|
|
30
|
+
with urllib.request.urlopen(req, timeout=timeout) as response:
|
|
31
|
+
result = json.loads(response.read().decode("utf-8"))
|
|
32
|
+
|
|
33
|
+
raw_response = result.get("response", "").strip()
|
|
34
|
+
try:
|
|
35
|
+
return json.loads(raw_response)
|
|
36
|
+
except json.JSONDecodeError:
|
|
37
|
+
return {"raw": raw_response}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Specialized domain definitions and model bindings for URI routing."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True)
|
|
8
|
+
class DomainProfile:
|
|
9
|
+
name: str
|
|
10
|
+
description: str
|
|
11
|
+
preferred_models: List[str]
|
|
12
|
+
uri_prefixes: List[str]
|
|
13
|
+
keywords: List[str]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
DOMAIN_REGISTRY: Dict[str, DomainProfile] = {
|
|
17
|
+
"code": DomainProfile(
|
|
18
|
+
name="code",
|
|
19
|
+
description="Software code synthesis, syntax verification, refactoring, and AST transformations.",
|
|
20
|
+
preferred_models=["qwen2.5-coder:3b", "willman-nlp:qwen2.5-3b"],
|
|
21
|
+
uri_prefixes=[
|
|
22
|
+
"willman://operation/code.",
|
|
23
|
+
"willman://operation/python.",
|
|
24
|
+
"willman://operation/git.resolve_conflicts",
|
|
25
|
+
"willman://operation/patch.",
|
|
26
|
+
],
|
|
27
|
+
keywords=["kod", "funkcja", "refaktoryzacja", "błąd", "python", "rust", "skrypt", "code", "bug", "patch"],
|
|
28
|
+
),
|
|
29
|
+
"api_ops": DomainProfile(
|
|
30
|
+
name="api_ops",
|
|
31
|
+
description="Structured API calls, GitHub issue tracking, Dockuri containers, Thunderbird mail, and filesystem operations.",
|
|
32
|
+
preferred_models=["willman-nlp:qwen2.5-3b", "granite4.1:3b"],
|
|
33
|
+
uri_prefixes=[
|
|
34
|
+
"willman://operation/github.",
|
|
35
|
+
"willman://operation/dockuri.",
|
|
36
|
+
"willman://operation/files.",
|
|
37
|
+
"willman://operation/thunderbird",
|
|
38
|
+
"dockuri://",
|
|
39
|
+
],
|
|
40
|
+
keywords=["github", "issue", "zgłoszenie", "plik", "dockuri", "kontener", "katalog", "read", "view", "email", "mail", "thunderbird", "poczta", "wiadomości"],
|
|
41
|
+
),
|
|
42
|
+
"planning": DomainProfile(
|
|
43
|
+
name="planning",
|
|
44
|
+
description="Sprint planning, ticket decomposition, subtask scheduling, and DAG dependencies.",
|
|
45
|
+
preferred_models=["willman-nlp:qwen2.5-3b", "qwen3.5:2b", "llama3.2:3b"],
|
|
46
|
+
uri_prefixes=[
|
|
47
|
+
"willman://operation/koru.",
|
|
48
|
+
"willman://operation/planfile.",
|
|
49
|
+
"willman://operation/schedule.",
|
|
50
|
+
],
|
|
51
|
+
keywords=["ticket", "zadanie", "plan", "sprint", "koru", "zależności", "podzadanie", "split", "board"],
|
|
52
|
+
),
|
|
53
|
+
"text": DomainProfile(
|
|
54
|
+
name="text",
|
|
55
|
+
description="Fast natural language text transformations, string normalization, whitespace, formatting.",
|
|
56
|
+
preferred_models=["willman-nlp:qwen2.5-3b", "willman-nlp:qwen2.5-1.5b"],
|
|
57
|
+
uri_prefixes=[
|
|
58
|
+
"willman://operation/text.",
|
|
59
|
+
],
|
|
60
|
+
keywords=["tekst", "odstępy", "wielkie", "litery", "normalize", "format", "string", "whitespace"],
|
|
61
|
+
),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def classify_domain(instruction: str, catalog_uris: Optional[List[str]] = None) -> DomainProfile:
|
|
66
|
+
"""Classify the target domain from natural language instruction and optional URI catalog."""
|
|
67
|
+
lowered = instruction.lower()
|
|
68
|
+
|
|
69
|
+
# Keyword and prefix matching score
|
|
70
|
+
best_domain = DOMAIN_REGISTRY["text"]
|
|
71
|
+
best_score = -1
|
|
72
|
+
|
|
73
|
+
for domain in DOMAIN_REGISTRY.values():
|
|
74
|
+
score = 0
|
|
75
|
+
for kw in domain.keywords:
|
|
76
|
+
if kw in lowered:
|
|
77
|
+
score += 2
|
|
78
|
+
if catalog_uris:
|
|
79
|
+
for uri in catalog_uris:
|
|
80
|
+
if any(uri.startswith(prefix) for prefix in domain.uri_prefixes):
|
|
81
|
+
score += 1
|
|
82
|
+
if score > best_score:
|
|
83
|
+
best_score = score
|
|
84
|
+
best_domain = domain
|
|
85
|
+
|
|
86
|
+
return best_domain
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def recommend_model(instruction: str, catalog_uris: Optional[List[str]] = None) -> str:
|
|
90
|
+
"""Recommend the best specialized model for an instruction."""
|
|
91
|
+
profile = classify_domain(instruction, catalog_uris)
|
|
92
|
+
return profile.preferred_models[0]
|
|
93
|
+
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Universal URI DSL for Paxlet / Willman ecosystem."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
|
5
|
+
from typing import Any, Dict, Tuple, Set
|
|
6
|
+
|
|
7
|
+
SUPPORTED_SCHEMES: Set[str] = {"willman", "proc", "dockuri", "paxlet"}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def with_params(uri: str, params: Dict[str, Any]) -> str:
|
|
11
|
+
"""Encode query parameters onto a base URI with deterministic JSON encoding."""
|
|
12
|
+
if urlsplit(uri).query:
|
|
13
|
+
raise ValueError("Base URI already has query parameters")
|
|
14
|
+
if not isinstance(params, dict):
|
|
15
|
+
raise ValueError("URI parameters must be a JSON object")
|
|
16
|
+
query = urlencode([
|
|
17
|
+
(str(key), json.dumps(value, ensure_ascii=False, separators=(",", ":")))
|
|
18
|
+
for key, value in sorted(params.items())
|
|
19
|
+
])
|
|
20
|
+
return uri + ("?" + query if query else "")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def parse_uri(uri: str) -> Tuple[str, Dict[str, Any]]:
|
|
24
|
+
"""Parse a URI into base URI and decoded parameter dictionary."""
|
|
25
|
+
parsed = urlsplit(uri)
|
|
26
|
+
if parsed.scheme not in SUPPORTED_SCHEMES or not parsed.netloc or parsed.fragment:
|
|
27
|
+
raise ValueError(f"Expected valid URI scheme in {SUPPORTED_SCHEMES} without fragment, got: {uri}")
|
|
28
|
+
values: Dict[str, Any] = {}
|
|
29
|
+
for key, raw in parse_qsl(parsed.query, keep_blank_values=True, strict_parsing=True):
|
|
30
|
+
if key in values:
|
|
31
|
+
raise ValueError(f"Duplicate URI parameter: {key}")
|
|
32
|
+
try:
|
|
33
|
+
values[key] = json.loads(raw)
|
|
34
|
+
except json.JSONDecodeError:
|
|
35
|
+
values[key] = raw
|
|
36
|
+
base = urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
|
|
37
|
+
return base, values
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def format_uri(scheme: str, entity_type: str, path: str, params: Dict[str, Any] | None = None) -> str:
|
|
41
|
+
"""Construct a canonical URI for an operation, container, or task."""
|
|
42
|
+
clean_path = path.lstrip("/")
|
|
43
|
+
base = f"{scheme}://{entity_type}/{clean_path}"
|
|
44
|
+
return with_params(base, params) if params else base
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def validate_uri(uri: str) -> bool:
|
|
48
|
+
"""Check whether a URI is syntactically valid."""
|
|
49
|
+
try:
|
|
50
|
+
parse_uri(uri)
|
|
51
|
+
return True
|
|
52
|
+
except Exception:
|
|
53
|
+
return False
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Central model catalog and configuration templates for small specialized LLMs."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True)
|
|
8
|
+
class ModelSpec:
|
|
9
|
+
name: str
|
|
10
|
+
family: str
|
|
11
|
+
parameters: str
|
|
12
|
+
vram_mb: int
|
|
13
|
+
context_tokens: int
|
|
14
|
+
domains: List[str]
|
|
15
|
+
description: str
|
|
16
|
+
system_prompt: str
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
MODEL_SPECS: Dict[str, ModelSpec] = {
|
|
20
|
+
"qwen2.5-coder:3b": ModelSpec(
|
|
21
|
+
name="qwen2.5-coder:3b",
|
|
22
|
+
family="qwen2.5",
|
|
23
|
+
parameters="3B",
|
|
24
|
+
vram_mb=2100,
|
|
25
|
+
context_tokens=8192,
|
|
26
|
+
domains=["code"],
|
|
27
|
+
description="High-precision code synthesis, AST analysis, Git conflicts, and automated patches.",
|
|
28
|
+
system_prompt="You are a code synthesis and AST transformation specialist. Return structured code or JSON.",
|
|
29
|
+
),
|
|
30
|
+
"willman-nlp:qwen2.5-3b": ModelSpec(
|
|
31
|
+
name="willman-nlp:qwen2.5-3b",
|
|
32
|
+
family="qwen2.5",
|
|
33
|
+
parameters="3B",
|
|
34
|
+
vram_mb=2100,
|
|
35
|
+
context_tokens=8192,
|
|
36
|
+
domains=["code", "api_ops", "planning", "text"],
|
|
37
|
+
description="Paxlet tuned generalist model for NL routing, URI parameter extraction, and operations.",
|
|
38
|
+
system_prompt="You are the Paxlet Willman NL-to-URI routing engine. Return exact JSON operations.",
|
|
39
|
+
),
|
|
40
|
+
"granite4.1:3b": ModelSpec(
|
|
41
|
+
name="granite4.1:3b",
|
|
42
|
+
family="granite",
|
|
43
|
+
parameters="3B",
|
|
44
|
+
vram_mb=1900,
|
|
45
|
+
context_tokens=8192,
|
|
46
|
+
domains=["api_ops"],
|
|
47
|
+
description="Enterprise tool-calling and structured API operations (GitHub, Dockuri, filesystem).",
|
|
48
|
+
system_prompt="You are an API operations specialist. Generate exact JSON API parameters and URI targets.",
|
|
49
|
+
),
|
|
50
|
+
"llama3.2:3b": ModelSpec(
|
|
51
|
+
name="llama3.2:3b",
|
|
52
|
+
family="llama",
|
|
53
|
+
parameters="3B",
|
|
54
|
+
vram_mb=2200,
|
|
55
|
+
context_tokens=8192,
|
|
56
|
+
domains=["planning"],
|
|
57
|
+
description="Sprint ticket decomposition, dependency DAG planning, and agile backlog management.",
|
|
58
|
+
system_prompt="You are a planning and sprint scheduling specialist. Break tasks into structured DAG steps.",
|
|
59
|
+
),
|
|
60
|
+
"qwen2.5:1.5b": ModelSpec(
|
|
61
|
+
name="qwen2.5:1.5b",
|
|
62
|
+
family="qwen2.5",
|
|
63
|
+
parameters="1.5B",
|
|
64
|
+
vram_mb=1100,
|
|
65
|
+
context_tokens=4096,
|
|
66
|
+
domains=["text"],
|
|
67
|
+
description="Ultra-lightweight text normalization, string formatting, and fast status checks.",
|
|
68
|
+
system_prompt="You are a fast text transformer. Clean, normalize, and format text strings concisely.",
|
|
69
|
+
),
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def get_model_for_domain(domain: str) -> str:
|
|
74
|
+
"""Return the primary recommended model for a given domain."""
|
|
75
|
+
for name, spec in MODEL_SPECS.items():
|
|
76
|
+
if domain in spec.domains:
|
|
77
|
+
return name
|
|
78
|
+
return "willman-nlp:qwen2.5-3b"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def generate_modelfile(model_name: str, context_tokens: int = 8192) -> str:
|
|
82
|
+
"""Generate an Ollama Modelfile content string for a model."""
|
|
83
|
+
spec = MODEL_SPECS.get(model_name)
|
|
84
|
+
base_from = model_name
|
|
85
|
+
system = spec.system_prompt if spec else "You are an autonomous engineering operations agent."
|
|
86
|
+
return f"""FROM {base_from}
|
|
87
|
+
PARAMETER num_ctx {context_tokens}
|
|
88
|
+
PARAMETER temperature 0.1
|
|
89
|
+
SYSTEM \"\"\"{system}\"\"\"
|
|
90
|
+
"""
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Standard system prompt templates and structured JSON schemas for willuri."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
from typing import Any, Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def build_routing_prompt(
|
|
8
|
+
domain_name: str,
|
|
9
|
+
catalog: List[Dict[str, Any]],
|
|
10
|
+
instruction: str,
|
|
11
|
+
input_data: Optional[Dict[str, Any]] = None,
|
|
12
|
+
) -> str:
|
|
13
|
+
"""Build a deterministic routing prompt for small LLMs."""
|
|
14
|
+
ops_lines = []
|
|
15
|
+
for op in catalog:
|
|
16
|
+
uri = op.get("uri", "")
|
|
17
|
+
desc = op.get("description", "")
|
|
18
|
+
ops_lines.append(f"- {uri}: {desc}")
|
|
19
|
+
catalog_str = "\n".join(ops_lines) if ops_lines else "No predefined operations; synthesize best URI."
|
|
20
|
+
|
|
21
|
+
return f"""You are a specialized URI Router for domain '{domain_name}'.
|
|
22
|
+
Available target process URIs:
|
|
23
|
+
{catalog_str}
|
|
24
|
+
|
|
25
|
+
User Instruction: {instruction}
|
|
26
|
+
Input context: {json.dumps(input_data or {}, ensure_ascii=False)}
|
|
27
|
+
|
|
28
|
+
Respond ONLY with a JSON object conforming to this schema:
|
|
29
|
+
{{
|
|
30
|
+
"uri": "exact target operation URI",
|
|
31
|
+
"input": {{ "param1": "value1" }},
|
|
32
|
+
"domain": "{domain_name}",
|
|
33
|
+
"reasoning": "brief justification"
|
|
34
|
+
}}
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def build_decomposition_prompt(ticket_title: str, ticket_description: str, catalog: List[Dict[str, Any]]) -> str:
|
|
39
|
+
"""Build a prompt for breaking down a high-level sprint ticket into discrete operations."""
|
|
40
|
+
ops_lines = [f"- {op.get('uri')}: {op.get('description', '')}" for op in catalog]
|
|
41
|
+
catalog_str = "\n".join(ops_lines)
|
|
42
|
+
|
|
43
|
+
return f"""You are a sprint ticket decomposition specialist.
|
|
44
|
+
Break down the following ticket into an ordered sequence of discrete URI operation steps.
|
|
45
|
+
|
|
46
|
+
Available operations:
|
|
47
|
+
{catalog_str}
|
|
48
|
+
|
|
49
|
+
Ticket: {ticket_title}
|
|
50
|
+
Description: {ticket_description}
|
|
51
|
+
|
|
52
|
+
Respond ONLY with a JSON object:
|
|
53
|
+
{{
|
|
54
|
+
"subtasks": [
|
|
55
|
+
{{
|
|
56
|
+
"title": "Subtask title",
|
|
57
|
+
"uri": "willman://operation/...",
|
|
58
|
+
"input": {{}}
|
|
59
|
+
}}
|
|
60
|
+
]
|
|
61
|
+
}}
|
|
62
|
+
"""
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""Core URI Router converting Natural Language into targeted URI operations."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
import time
|
|
5
|
+
from typing import Any, Dict, List, Optional
|
|
6
|
+
|
|
7
|
+
from .domains import classify_domain, DOMAIN_REGISTRY, DomainProfile
|
|
8
|
+
from .client import query_ollama
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class UriRouter:
|
|
12
|
+
"""Specialized NL-to-URI process router selecting domain-specific models."""
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
catalog: Optional[List[Dict[str, Any]]] = None,
|
|
17
|
+
ollama_host: str = "http://127.0.0.1:11434",
|
|
18
|
+
provider: Optional[Any] = None,
|
|
19
|
+
):
|
|
20
|
+
self.catalog = list(catalog or [])
|
|
21
|
+
self.ollama_host = ollama_host
|
|
22
|
+
self.provider = provider or query_ollama
|
|
23
|
+
|
|
24
|
+
def add_operation(self, uri: str, description: str, input_schema: Optional[Dict[str, Any]] = None):
|
|
25
|
+
self.catalog.append({
|
|
26
|
+
"uri": uri,
|
|
27
|
+
"description": description,
|
|
28
|
+
"input_schema": input_schema or {},
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
def route(
|
|
32
|
+
self,
|
|
33
|
+
instruction: str,
|
|
34
|
+
input_data: Optional[Dict[str, Any]] = None,
|
|
35
|
+
model_override: Optional[str] = None,
|
|
36
|
+
timeout: float = 20.0,
|
|
37
|
+
) -> Dict[str, Any]:
|
|
38
|
+
"""Route an NL instruction to the best URI operation using a domain-specialized small LLM."""
|
|
39
|
+
start_time = time.monotonic()
|
|
40
|
+
|
|
41
|
+
# AC: empty catalog unsupported
|
|
42
|
+
if not self.catalog:
|
|
43
|
+
return {
|
|
44
|
+
"status": "unsupported",
|
|
45
|
+
"error": "empty_catalog",
|
|
46
|
+
"message": "Empty catalog: routing unsupported without registered operations",
|
|
47
|
+
"uri": None,
|
|
48
|
+
"input": {},
|
|
49
|
+
"domain": None,
|
|
50
|
+
"model_used": None,
|
|
51
|
+
"duration_ms": 0.0,
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
catalog_uris = [op["uri"] for op in self.catalog]
|
|
55
|
+
domain: DomainProfile = classify_domain(instruction, catalog_uris)
|
|
56
|
+
selected_model = model_override or domain.preferred_models[0]
|
|
57
|
+
|
|
58
|
+
# Build compact catalog prompt
|
|
59
|
+
ops_text = []
|
|
60
|
+
for op in self.catalog:
|
|
61
|
+
schema_desc = f" (schema: {json.dumps(op.get('input_schema', {}))})" if op.get("input_schema") else ""
|
|
62
|
+
ops_text.append(f"- {op['uri']}: {op.get('description', '')}{schema_desc}")
|
|
63
|
+
catalog_str = "\n".join(ops_text)
|
|
64
|
+
|
|
65
|
+
prompt = f"""You are a specialized URI Router for domain '{domain.name}'.
|
|
66
|
+
Given the following available registered process URIs:
|
|
67
|
+
{catalog_str}
|
|
68
|
+
|
|
69
|
+
User Instruction: {instruction}
|
|
70
|
+
Input context: {json.dumps(input_data or {}, ensure_ascii=False)}
|
|
71
|
+
|
|
72
|
+
Respond ONLY with a JSON object containing:
|
|
73
|
+
- "uri": exact matched registered process URI from the list above, or null if no registered URI matches
|
|
74
|
+
- "input": dictionary of extracted arguments matching the operation schema
|
|
75
|
+
- "domain": "{domain.name}"
|
|
76
|
+
"""
|
|
77
|
+
try:
|
|
78
|
+
response = self.provider(
|
|
79
|
+
model=selected_model,
|
|
80
|
+
prompt=prompt,
|
|
81
|
+
timeout=timeout,
|
|
82
|
+
host=self.ollama_host,
|
|
83
|
+
)
|
|
84
|
+
except Exception as ex:
|
|
85
|
+
duration_ms = round((time.monotonic() - start_time) * 1000, 2)
|
|
86
|
+
return {
|
|
87
|
+
"status": "unsupported",
|
|
88
|
+
"error": "provider_unavailable",
|
|
89
|
+
"message": f"LLM provider error: {ex}",
|
|
90
|
+
"uri": None,
|
|
91
|
+
"input": {},
|
|
92
|
+
"domain": domain.name,
|
|
93
|
+
"model_used": selected_model,
|
|
94
|
+
"duration_ms": duration_ms,
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
duration_ms = round((time.monotonic() - start_time) * 1000, 2)
|
|
98
|
+
|
|
99
|
+
matched_uri = response.get("uri")
|
|
100
|
+
if not matched_uri:
|
|
101
|
+
return {
|
|
102
|
+
"status": "unsupported",
|
|
103
|
+
"error": "no_matched_uri",
|
|
104
|
+
"message": "Instruction could not be resolved to any registered URI",
|
|
105
|
+
"uri": None,
|
|
106
|
+
"input": {},
|
|
107
|
+
"domain": domain.name,
|
|
108
|
+
"model_used": selected_model,
|
|
109
|
+
"duration_ms": duration_ms,
|
|
110
|
+
"raw": response,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
# AC: only selected registered URI yields success; reject invented operations
|
|
114
|
+
target_op = next((op for op in self.catalog if op.get("uri") == matched_uri), None)
|
|
115
|
+
if not target_op:
|
|
116
|
+
return {
|
|
117
|
+
"status": "unsupported",
|
|
118
|
+
"error": "unregistered_uri",
|
|
119
|
+
"message": f"Operation '{matched_uri}' is not registered in catalog",
|
|
120
|
+
"uri": None,
|
|
121
|
+
"input": {},
|
|
122
|
+
"domain": domain.name,
|
|
123
|
+
"model_used": selected_model,
|
|
124
|
+
"duration_ms": duration_ms,
|
|
125
|
+
"raw": response,
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
input_args = response.get("input", {})
|
|
129
|
+
schema = target_op.get("input_schema")
|
|
130
|
+
if schema:
|
|
131
|
+
try:
|
|
132
|
+
import jsonschema
|
|
133
|
+
jsonschema.validate(instance=input_args, schema=schema)
|
|
134
|
+
except Exception as ve:
|
|
135
|
+
return {
|
|
136
|
+
"status": "clarify",
|
|
137
|
+
"error": "invalid_arguments",
|
|
138
|
+
"message": f"Argument validation against schema failed: {ve}",
|
|
139
|
+
"uri": target_op["uri"],
|
|
140
|
+
"input": input_args,
|
|
141
|
+
"domain": domain.name,
|
|
142
|
+
"model_used": selected_model,
|
|
143
|
+
"duration_ms": duration_ms,
|
|
144
|
+
"raw": response,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
return {
|
|
148
|
+
"status": "ok",
|
|
149
|
+
"uri": target_op["uri"],
|
|
150
|
+
"input": input_args,
|
|
151
|
+
"domain": domain.name,
|
|
152
|
+
"model_used": selected_model,
|
|
153
|
+
"duration_ms": duration_ms,
|
|
154
|
+
"raw": response,
|
|
155
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Fast local lexical matcher and candidate ranker for operations.
|
|
2
|
+
|
|
3
|
+
Uses Jaccard token overlap for lexical keyword pre-filtering across catalog operations.
|
|
4
|
+
Note: For semantic vector retrieval, use an embedding adapter rather than lexical overlap.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
import math
|
|
8
|
+
import re
|
|
9
|
+
from typing import Any, Dict, List, Tuple
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def tokenize(text: str) -> List[str]:
|
|
13
|
+
"""Tokenize text into lowercase alphanumeric keywords."""
|
|
14
|
+
return [t for t in re.findall(r"\w+", text.lower()) if len(t) > 1]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def compute_similarity(tokens_a: List[str], tokens_b: List[str]) -> float:
|
|
18
|
+
"""Compute Jaccard token similarity with term frequency weighting."""
|
|
19
|
+
if not tokens_a or not tokens_b:
|
|
20
|
+
return 0.0
|
|
21
|
+
set_a = set(tokens_a)
|
|
22
|
+
set_b = set(tokens_b)
|
|
23
|
+
intersection = set_a.intersection(set_b)
|
|
24
|
+
union = set_a.union(set_b)
|
|
25
|
+
return len(intersection) / len(union) if union else 0.0
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SemanticMatcher:
|
|
29
|
+
"""Ranks and pre-filters catalog operations against natural language instructions."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, catalog: List[Dict[str, Any]]):
|
|
32
|
+
self.catalog = catalog
|
|
33
|
+
|
|
34
|
+
def match(self, instruction: str, top_k: int = 5) -> List[Tuple[Dict[str, Any], float]]:
|
|
35
|
+
"""Return the top_k most relevant operations with relevance scores."""
|
|
36
|
+
inst_tokens = tokenize(instruction)
|
|
37
|
+
scored: List[Tuple[Dict[str, Any], float]] = []
|
|
38
|
+
|
|
39
|
+
for op in self.catalog:
|
|
40
|
+
text = f"{op.get('uri', '')} {op.get('description', '')} {' '.join(op.get('aliases', {}).keys())}"
|
|
41
|
+
op_tokens = tokenize(text)
|
|
42
|
+
score = compute_similarity(inst_tokens, op_tokens)
|
|
43
|
+
if score > 0.0:
|
|
44
|
+
scored.append((op, round(score, 4)))
|
|
45
|
+
|
|
46
|
+
scored.sort(key=lambda x: x[1], reverse=True)
|
|
47
|
+
return scored[:top_k]
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: willuri
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Natural language to specialized URI process router with domain-specific small LLMs
|
|
5
|
+
Author-email: Tom Sapletta <tom@sapletta.com>
|
|
6
|
+
License: Proprietary
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: pyyaml>=6.0
|
|
11
|
+
Dynamic: license-file
|
|
12
|
+
|
|
13
|
+
# willuri
|
|
14
|
+
|
|
15
|
+
Natural language to specialized process URI router powered by domain-specific lightweight LLMs.
|
|
16
|
+
|
|
17
|
+
## Overview
|
|
18
|
+
|
|
19
|
+
In the Paxlet / Willman ecosystem, every capability is addressed by an explicit process URI (`willman://operation/...`, `dockuri://...`).
|
|
20
|
+
`willuri` maps natural language user instructions or tickets from the board to specialized process URIs using small, fast, domain-targeted LLMs instead of a monolithic general-purpose model.
|
|
21
|
+
|
|
22
|
+
### Specialized Domain Models
|
|
23
|
+
|
|
24
|
+
| Domain | Process URI Scope | Primary Model | Alternative | Focus |
|
|
25
|
+
| :--- | :--- | :--- | :--- | :--- |
|
|
26
|
+
| **`code`** | `willman://operation/code.*`, `git.resolve_conflicts` | `qwen2.5-coder:3b` | `willman-nlp:qwen2.5-3b` | Code synthesis, AST edits, diffs, bug repairs |
|
|
27
|
+
| **`api_ops`** | `willman://operation/github.*`, `dockuri.*`, `files.*` | `granite4.1:3b` | `willman-nlp:qwen2.5-3b` | Structured API tools, JSON schema, enterprise ops |
|
|
28
|
+
| **`planning`** | `willman://operation/koru.*`, `planfile.*`, `schedule.*` | `llama3.2:3b` | `willman-nlp:qwen2.5-3b` | Ticket decomposition, sprint planning, DAG subtasks |
|
|
29
|
+
| **`text`** | `willman://operation/text.*` | `willman-nlp:qwen2.5-1.5b` | `willman-nlp:qwen2.5-3b` | String normalization, fast formatting, prose |
|
|
30
|
+
|
|
31
|
+
## Usage
|
|
32
|
+
|
|
33
|
+
### Python API
|
|
34
|
+
```python
|
|
35
|
+
from willuri.router import UriRouter
|
|
36
|
+
|
|
37
|
+
router = UriRouter()
|
|
38
|
+
router.add_operation("willman://operation/text.normalize/v1", "Normalize whitespace")
|
|
39
|
+
router.add_operation("willman://operation/git.resolve_conflicts/v1", "Resolve git merge conflicts")
|
|
40
|
+
|
|
41
|
+
result = router.route("Rozwiąż konflikty git w repozytorium")
|
|
42
|
+
print(result["uri"])
|
|
43
|
+
# => "willman://operation/git.resolve_conflicts/v1"
|
|
44
|
+
print(result["model_used"])
|
|
45
|
+
# => "qwen2.5-coder:3b"
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### CLI
|
|
49
|
+
```bash
|
|
50
|
+
willuri classify "Popraw błąd w funkcji liczącej sumę"
|
|
51
|
+
willuri route "Znormalizuj odstępy w tekście 'Ala ma kota'"
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Verification
|
|
55
|
+
```bash
|
|
56
|
+
python3 -m unittest discover -s tests -v
|
|
57
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
tests/test_dsl_and_models.py
|
|
5
|
+
tests/test_router.py
|
|
6
|
+
tests/test_ssot_config.py
|
|
7
|
+
willuri/__init__.py
|
|
8
|
+
willuri/cli.py
|
|
9
|
+
willuri/client.py
|
|
10
|
+
willuri/domains.py
|
|
11
|
+
willuri/dsl.py
|
|
12
|
+
willuri/models.py
|
|
13
|
+
willuri/prompts.py
|
|
14
|
+
willuri/router.py
|
|
15
|
+
willuri/semantic.py
|
|
16
|
+
willuri.egg-info/PKG-INFO
|
|
17
|
+
willuri.egg-info/SOURCES.txt
|
|
18
|
+
willuri.egg-info/dependency_links.txt
|
|
19
|
+
willuri.egg-info/entry_points.txt
|
|
20
|
+
willuri.egg-info/requires.txt
|
|
21
|
+
willuri.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pyyaml>=6.0
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
willuri
|