willuri 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
willuri/__init__.py ADDED
@@ -0,0 +1,29 @@
1
+ """willuri: Universal NL, URI, and specialized LLM router SSOT for Paxlet / Willman."""
2
+ from .domains import DOMAIN_REGISTRY, DomainProfile, classify_domain
3
+ from .router import UriRouter
4
+ from .client import query_ollama
5
+ from .dsl import parse_uri, with_params, format_uri, validate_uri
6
+ from .models import MODEL_SPECS, get_model_for_domain, generate_modelfile
7
+ from .prompts import build_routing_prompt, build_decomposition_prompt
8
+ from .semantic import SemanticMatcher, compute_similarity, tokenize
9
+
10
+ __version__ = "0.1.0"
11
+ __all__ = [
12
+ "UriRouter",
13
+ "classify_domain",
14
+ "DOMAIN_REGISTRY",
15
+ "DomainProfile",
16
+ "query_ollama",
17
+ "parse_uri",
18
+ "with_params",
19
+ "format_uri",
20
+ "validate_uri",
21
+ "MODEL_SPECS",
22
+ "get_model_for_domain",
23
+ "generate_modelfile",
24
+ "build_routing_prompt",
25
+ "build_decomposition_prompt",
26
+ "SemanticMatcher",
27
+ "compute_similarity",
28
+ "tokenize",
29
+ ]
willuri/cli.py ADDED
@@ -0,0 +1,58 @@
1
+ """CLI for willuri router."""
2
+ import argparse
3
+ import json
4
+ import sys
5
+ from .router import UriRouter
6
+ from .domains import classify_domain
7
+
8
+
9
+ def main():
10
+ parser = argparse.ArgumentParser(prog="willuri", description="Willuri: NL to URI Process Router")
11
+ sub = parser.add_subparsers(dest="command", required=True)
12
+
13
+ sub.add_parser("domains", help="List all registered domains and specialized models")
14
+
15
+ classify_cmd = sub.add_parser("classify", help="Classify instruction domain")
16
+ classify_cmd.add_argument("instruction", help="Natural language instruction")
17
+
18
+ route_cmd = sub.add_parser("route", help="Route instruction to URI via specialized LLM")
19
+ route_cmd.add_argument("instruction", help="Natural language instruction")
20
+ route_cmd.add_argument("--model", help="Override LLM model")
21
+ route_cmd.add_argument("--input", default="{}", help="Input JSON data")
22
+
23
+ args = parser.parse_args()
24
+
25
+ if args.command == "domains":
26
+ from .domains import DOMAIN_REGISTRY
27
+ res = {
28
+ name: {
29
+ "description": prof.description,
30
+ "preferred_models": prof.preferred_models,
31
+ "uri_prefixes": prof.uri_prefixes,
32
+ "keywords": prof.keywords,
33
+ }
34
+ for name, prof in DOMAIN_REGISTRY.items()
35
+ }
36
+ print(json.dumps(res, indent=2, ensure_ascii=False))
37
+ return 0
38
+
39
+ if args.command == "classify":
40
+ domain = classify_domain(args.instruction)
41
+ print(json.dumps({
42
+ "domain": domain.name,
43
+ "description": domain.description,
44
+ "preferred_models": domain.preferred_models,
45
+ "uri_prefixes": domain.uri_prefixes,
46
+ }, indent=2, ensure_ascii=False))
47
+ return 0
48
+
49
+ if args.command == "route":
50
+ input_data = json.loads(args.input)
51
+ router = UriRouter()
52
+ result = router.route(args.instruction, input_data=input_data, model_override=args.model)
53
+ print(json.dumps(result, indent=2, ensure_ascii=False))
54
+ return 0 if result.get("status") in ("ok", "routed") else 1
55
+
56
+
57
+ if __name__ == "__main__":
58
+ sys.exit(main())
willuri/client.py ADDED
@@ -0,0 +1,37 @@
1
+ """Specialized LLM inference client for URI parameter extraction and routing."""
2
+ from __future__ import annotations
3
+ import json
4
+ import os
5
+ import urllib.request
6
+ from typing import Any, Dict, List, Optional
7
+
8
+
9
+ def query_ollama(
10
+ model: str,
11
+ prompt: str,
12
+ schema: Optional[Dict[str, Any]] = None,
13
+ timeout: float = 30.0,
14
+ host: str = "http://127.0.0.1:11434",
15
+ ) -> Dict[str, Any]:
16
+ """Execute a structured generation query against local Ollama endpoint."""
17
+ url = f"{host.rstrip('/')}/api/generate"
18
+ payload: Dict[str, Any] = {
19
+ "model": model,
20
+ "prompt": prompt,
21
+ "stream": False,
22
+ }
23
+ if schema:
24
+ payload["format"] = schema
25
+ else:
26
+ payload["format"] = "json"
27
+
28
+ data = json.dumps(payload).encode("utf-8")
29
+ req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json"})
30
+ with urllib.request.urlopen(req, timeout=timeout) as response:
31
+ result = json.loads(response.read().decode("utf-8"))
32
+
33
+ raw_response = result.get("response", "").strip()
34
+ try:
35
+ return json.loads(raw_response)
36
+ except json.JSONDecodeError:
37
+ return {"raw": raw_response}
willuri/domains.py ADDED
@@ -0,0 +1,93 @@
1
+ """Specialized domain definitions and model bindings for URI routing."""
2
+ from __future__ import annotations
3
+ from dataclasses import dataclass
4
+ from typing import Dict, List, Optional
5
+
6
+
7
+ @dataclass(frozen=True)
8
+ class DomainProfile:
9
+ name: str
10
+ description: str
11
+ preferred_models: List[str]
12
+ uri_prefixes: List[str]
13
+ keywords: List[str]
14
+
15
+
16
+ DOMAIN_REGISTRY: Dict[str, DomainProfile] = {
17
+ "code": DomainProfile(
18
+ name="code",
19
+ description="Software code synthesis, syntax verification, refactoring, and AST transformations.",
20
+ preferred_models=["qwen2.5-coder:3b", "willman-nlp:qwen2.5-3b"],
21
+ uri_prefixes=[
22
+ "willman://operation/code.",
23
+ "willman://operation/python.",
24
+ "willman://operation/git.resolve_conflicts",
25
+ "willman://operation/patch.",
26
+ ],
27
+ keywords=["kod", "funkcja", "refaktoryzacja", "błąd", "python", "rust", "skrypt", "code", "bug", "patch"],
28
+ ),
29
+ "api_ops": DomainProfile(
30
+ name="api_ops",
31
+ description="Structured API calls, GitHub issue tracking, Dockuri containers, Thunderbird mail, and filesystem operations.",
32
+ preferred_models=["willman-nlp:qwen2.5-3b", "granite4.1:3b"],
33
+ uri_prefixes=[
34
+ "willman://operation/github.",
35
+ "willman://operation/dockuri.",
36
+ "willman://operation/files.",
37
+ "willman://operation/thunderbird",
38
+ "dockuri://",
39
+ ],
40
+ keywords=["github", "issue", "zgłoszenie", "plik", "dockuri", "kontener", "katalog", "read", "view", "email", "mail", "thunderbird", "poczta", "wiadomości"],
41
+ ),
42
+ "planning": DomainProfile(
43
+ name="planning",
44
+ description="Sprint planning, ticket decomposition, subtask scheduling, and DAG dependencies.",
45
+ preferred_models=["willman-nlp:qwen2.5-3b", "qwen3.5:2b", "llama3.2:3b"],
46
+ uri_prefixes=[
47
+ "willman://operation/koru.",
48
+ "willman://operation/planfile.",
49
+ "willman://operation/schedule.",
50
+ ],
51
+ keywords=["ticket", "zadanie", "plan", "sprint", "koru", "zależności", "podzadanie", "split", "board"],
52
+ ),
53
+ "text": DomainProfile(
54
+ name="text",
55
+ description="Fast natural language text transformations, string normalization, whitespace, formatting.",
56
+ preferred_models=["willman-nlp:qwen2.5-3b", "willman-nlp:qwen2.5-1.5b"],
57
+ uri_prefixes=[
58
+ "willman://operation/text.",
59
+ ],
60
+ keywords=["tekst", "odstępy", "wielkie", "litery", "normalize", "format", "string", "whitespace"],
61
+ ),
62
+ }
63
+
64
+
65
+ def classify_domain(instruction: str, catalog_uris: Optional[List[str]] = None) -> DomainProfile:
66
+ """Classify the target domain from natural language instruction and optional URI catalog."""
67
+ lowered = instruction.lower()
68
+
69
+ # Keyword and prefix matching score
70
+ best_domain = DOMAIN_REGISTRY["text"]
71
+ best_score = -1
72
+
73
+ for domain in DOMAIN_REGISTRY.values():
74
+ score = 0
75
+ for kw in domain.keywords:
76
+ if kw in lowered:
77
+ score += 2
78
+ if catalog_uris:
79
+ for uri in catalog_uris:
80
+ if any(uri.startswith(prefix) for prefix in domain.uri_prefixes):
81
+ score += 1
82
+ if score > best_score:
83
+ best_score = score
84
+ best_domain = domain
85
+
86
+ return best_domain
87
+
88
+
89
+ def recommend_model(instruction: str, catalog_uris: Optional[List[str]] = None) -> str:
90
+ """Recommend the best specialized model for an instruction."""
91
+ profile = classify_domain(instruction, catalog_uris)
92
+ return profile.preferred_models[0]
93
+
willuri/dsl.py ADDED
@@ -0,0 +1,53 @@
1
+ """Universal URI DSL for Paxlet / Willman ecosystem."""
2
+ from __future__ import annotations
3
+ import json
4
+ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
5
+ from typing import Any, Dict, Tuple, Set
6
+
7
+ SUPPORTED_SCHEMES: Set[str] = {"willman", "proc", "dockuri", "paxlet"}
8
+
9
+
10
+ def with_params(uri: str, params: Dict[str, Any]) -> str:
11
+ """Encode query parameters onto a base URI with deterministic JSON encoding."""
12
+ if urlsplit(uri).query:
13
+ raise ValueError("Base URI already has query parameters")
14
+ if not isinstance(params, dict):
15
+ raise ValueError("URI parameters must be a JSON object")
16
+ query = urlencode([
17
+ (str(key), json.dumps(value, ensure_ascii=False, separators=(",", ":")))
18
+ for key, value in sorted(params.items())
19
+ ])
20
+ return uri + ("?" + query if query else "")
21
+
22
+
23
+ def parse_uri(uri: str) -> Tuple[str, Dict[str, Any]]:
24
+ """Parse a URI into base URI and decoded parameter dictionary."""
25
+ parsed = urlsplit(uri)
26
+ if parsed.scheme not in SUPPORTED_SCHEMES or not parsed.netloc or parsed.fragment:
27
+ raise ValueError(f"Expected valid URI scheme in {SUPPORTED_SCHEMES} without fragment, got: {uri}")
28
+ values: Dict[str, Any] = {}
29
+ for key, raw in parse_qsl(parsed.query, keep_blank_values=True, strict_parsing=True):
30
+ if key in values:
31
+ raise ValueError(f"Duplicate URI parameter: {key}")
32
+ try:
33
+ values[key] = json.loads(raw)
34
+ except json.JSONDecodeError:
35
+ values[key] = raw
36
+ base = urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
37
+ return base, values
38
+
39
+
40
+ def format_uri(scheme: str, entity_type: str, path: str, params: Dict[str, Any] | None = None) -> str:
41
+ """Construct a canonical URI for an operation, container, or task."""
42
+ clean_path = path.lstrip("/")
43
+ base = f"{scheme}://{entity_type}/{clean_path}"
44
+ return with_params(base, params) if params else base
45
+
46
+
47
+ def validate_uri(uri: str) -> bool:
48
+ """Check whether a URI is syntactically valid."""
49
+ try:
50
+ parse_uri(uri)
51
+ return True
52
+ except Exception:
53
+ return False
willuri/models.py ADDED
@@ -0,0 +1,90 @@
1
+ """Central model catalog and configuration templates for small specialized LLMs."""
2
+ from __future__ import annotations
3
+ from dataclasses import dataclass
4
+ from typing import Dict, List, Optional
5
+
6
+
7
+ @dataclass(frozen=True)
8
+ class ModelSpec:
9
+ name: str
10
+ family: str
11
+ parameters: str
12
+ vram_mb: int
13
+ context_tokens: int
14
+ domains: List[str]
15
+ description: str
16
+ system_prompt: str
17
+
18
+
19
+ MODEL_SPECS: Dict[str, ModelSpec] = {
20
+ "qwen2.5-coder:3b": ModelSpec(
21
+ name="qwen2.5-coder:3b",
22
+ family="qwen2.5",
23
+ parameters="3B",
24
+ vram_mb=2100,
25
+ context_tokens=8192,
26
+ domains=["code"],
27
+ description="High-precision code synthesis, AST analysis, Git conflicts, and automated patches.",
28
+ system_prompt="You are a code synthesis and AST transformation specialist. Return structured code or JSON.",
29
+ ),
30
+ "willman-nlp:qwen2.5-3b": ModelSpec(
31
+ name="willman-nlp:qwen2.5-3b",
32
+ family="qwen2.5",
33
+ parameters="3B",
34
+ vram_mb=2100,
35
+ context_tokens=8192,
36
+ domains=["code", "api_ops", "planning", "text"],
37
+ description="Paxlet tuned generalist model for NL routing, URI parameter extraction, and operations.",
38
+ system_prompt="You are the Paxlet Willman NL-to-URI routing engine. Return exact JSON operations.",
39
+ ),
40
+ "granite4.1:3b": ModelSpec(
41
+ name="granite4.1:3b",
42
+ family="granite",
43
+ parameters="3B",
44
+ vram_mb=1900,
45
+ context_tokens=8192,
46
+ domains=["api_ops"],
47
+ description="Enterprise tool-calling and structured API operations (GitHub, Dockuri, filesystem).",
48
+ system_prompt="You are an API operations specialist. Generate exact JSON API parameters and URI targets.",
49
+ ),
50
+ "llama3.2:3b": ModelSpec(
51
+ name="llama3.2:3b",
52
+ family="llama",
53
+ parameters="3B",
54
+ vram_mb=2200,
55
+ context_tokens=8192,
56
+ domains=["planning"],
57
+ description="Sprint ticket decomposition, dependency DAG planning, and agile backlog management.",
58
+ system_prompt="You are a planning and sprint scheduling specialist. Break tasks into structured DAG steps.",
59
+ ),
60
+ "qwen2.5:1.5b": ModelSpec(
61
+ name="qwen2.5:1.5b",
62
+ family="qwen2.5",
63
+ parameters="1.5B",
64
+ vram_mb=1100,
65
+ context_tokens=4096,
66
+ domains=["text"],
67
+ description="Ultra-lightweight text normalization, string formatting, and fast status checks.",
68
+ system_prompt="You are a fast text transformer. Clean, normalize, and format text strings concisely.",
69
+ ),
70
+ }
71
+
72
+
73
+ def get_model_for_domain(domain: str) -> str:
74
+ """Return the primary recommended model for a given domain."""
75
+ for name, spec in MODEL_SPECS.items():
76
+ if domain in spec.domains:
77
+ return name
78
+ return "willman-nlp:qwen2.5-3b"
79
+
80
+
81
+ def generate_modelfile(model_name: str, context_tokens: int = 8192) -> str:
82
+ """Generate an Ollama Modelfile content string for a model."""
83
+ spec = MODEL_SPECS.get(model_name)
84
+ base_from = model_name
85
+ system = spec.system_prompt if spec else "You are an autonomous engineering operations agent."
86
+ return f"""FROM {base_from}
87
+ PARAMETER num_ctx {context_tokens}
88
+ PARAMETER temperature 0.1
89
+ SYSTEM \"\"\"{system}\"\"\"
90
+ """
willuri/prompts.py ADDED
@@ -0,0 +1,62 @@
1
+ """Standard system prompt templates and structured JSON schemas for willuri."""
2
+ from __future__ import annotations
3
+ import json
4
+ from typing import Any, Dict, List, Optional
5
+
6
+
7
+ def build_routing_prompt(
8
+ domain_name: str,
9
+ catalog: List[Dict[str, Any]],
10
+ instruction: str,
11
+ input_data: Optional[Dict[str, Any]] = None,
12
+ ) -> str:
13
+ """Build a deterministic routing prompt for small LLMs."""
14
+ ops_lines = []
15
+ for op in catalog:
16
+ uri = op.get("uri", "")
17
+ desc = op.get("description", "")
18
+ ops_lines.append(f"- {uri}: {desc}")
19
+ catalog_str = "\n".join(ops_lines) if ops_lines else "No predefined operations; synthesize best URI."
20
+
21
+ return f"""You are a specialized URI Router for domain '{domain_name}'.
22
+ Available target process URIs:
23
+ {catalog_str}
24
+
25
+ User Instruction: {instruction}
26
+ Input context: {json.dumps(input_data or {}, ensure_ascii=False)}
27
+
28
+ Respond ONLY with a JSON object conforming to this schema:
29
+ {{
30
+ "uri": "exact target operation URI",
31
+ "input": {{ "param1": "value1" }},
32
+ "domain": "{domain_name}",
33
+ "reasoning": "brief justification"
34
+ }}
35
+ """
36
+
37
+
38
+ def build_decomposition_prompt(ticket_title: str, ticket_description: str, catalog: List[Dict[str, Any]]) -> str:
39
+ """Build a prompt for breaking down a high-level sprint ticket into discrete operations."""
40
+ ops_lines = [f"- {op.get('uri')}: {op.get('description', '')}" for op in catalog]
41
+ catalog_str = "\n".join(ops_lines)
42
+
43
+ return f"""You are a sprint ticket decomposition specialist.
44
+ Break down the following ticket into an ordered sequence of discrete URI operation steps.
45
+
46
+ Available operations:
47
+ {catalog_str}
48
+
49
+ Ticket: {ticket_title}
50
+ Description: {ticket_description}
51
+
52
+ Respond ONLY with a JSON object:
53
+ {{
54
+ "subtasks": [
55
+ {{
56
+ "title": "Subtask title",
57
+ "uri": "willman://operation/...",
58
+ "input": {{}}
59
+ }}
60
+ ]
61
+ }}
62
+ """
willuri/router.py ADDED
@@ -0,0 +1,155 @@
1
+ """Core URI Router converting Natural Language into targeted URI operations."""
2
+ from __future__ import annotations
3
+ import json
4
+ import time
5
+ from typing import Any, Dict, List, Optional
6
+
7
+ from .domains import classify_domain, DOMAIN_REGISTRY, DomainProfile
8
+ from .client import query_ollama
9
+
10
+
11
+ class UriRouter:
12
+ """Specialized NL-to-URI process router selecting domain-specific models."""
13
+
14
+ def __init__(
15
+ self,
16
+ catalog: Optional[List[Dict[str, Any]]] = None,
17
+ ollama_host: str = "http://127.0.0.1:11434",
18
+ provider: Optional[Any] = None,
19
+ ):
20
+ self.catalog = list(catalog or [])
21
+ self.ollama_host = ollama_host
22
+ self.provider = provider or query_ollama
23
+
24
+ def add_operation(self, uri: str, description: str, input_schema: Optional[Dict[str, Any]] = None):
25
+ self.catalog.append({
26
+ "uri": uri,
27
+ "description": description,
28
+ "input_schema": input_schema or {},
29
+ })
30
+
31
+ def route(
32
+ self,
33
+ instruction: str,
34
+ input_data: Optional[Dict[str, Any]] = None,
35
+ model_override: Optional[str] = None,
36
+ timeout: float = 20.0,
37
+ ) -> Dict[str, Any]:
38
+ """Route an NL instruction to the best URI operation using a domain-specialized small LLM."""
39
+ start_time = time.monotonic()
40
+
41
+ # AC: empty catalog unsupported
42
+ if not self.catalog:
43
+ return {
44
+ "status": "unsupported",
45
+ "error": "empty_catalog",
46
+ "message": "Empty catalog: routing unsupported without registered operations",
47
+ "uri": None,
48
+ "input": {},
49
+ "domain": None,
50
+ "model_used": None,
51
+ "duration_ms": 0.0,
52
+ }
53
+
54
+ catalog_uris = [op["uri"] for op in self.catalog]
55
+ domain: DomainProfile = classify_domain(instruction, catalog_uris)
56
+ selected_model = model_override or domain.preferred_models[0]
57
+
58
+ # Build compact catalog prompt
59
+ ops_text = []
60
+ for op in self.catalog:
61
+ schema_desc = f" (schema: {json.dumps(op.get('input_schema', {}))})" if op.get("input_schema") else ""
62
+ ops_text.append(f"- {op['uri']}: {op.get('description', '')}{schema_desc}")
63
+ catalog_str = "\n".join(ops_text)
64
+
65
+ prompt = f"""You are a specialized URI Router for domain '{domain.name}'.
66
+ Given the following available registered process URIs:
67
+ {catalog_str}
68
+
69
+ User Instruction: {instruction}
70
+ Input context: {json.dumps(input_data or {}, ensure_ascii=False)}
71
+
72
+ Respond ONLY with a JSON object containing:
73
+ - "uri": exact matched registered process URI from the list above, or null if no registered URI matches
74
+ - "input": dictionary of extracted arguments matching the operation schema
75
+ - "domain": "{domain.name}"
76
+ """
77
+ try:
78
+ response = self.provider(
79
+ model=selected_model,
80
+ prompt=prompt,
81
+ timeout=timeout,
82
+ host=self.ollama_host,
83
+ )
84
+ except Exception as ex:
85
+ duration_ms = round((time.monotonic() - start_time) * 1000, 2)
86
+ return {
87
+ "status": "unsupported",
88
+ "error": "provider_unavailable",
89
+ "message": f"LLM provider error: {ex}",
90
+ "uri": None,
91
+ "input": {},
92
+ "domain": domain.name,
93
+ "model_used": selected_model,
94
+ "duration_ms": duration_ms,
95
+ }
96
+
97
+ duration_ms = round((time.monotonic() - start_time) * 1000, 2)
98
+
99
+ matched_uri = response.get("uri")
100
+ if not matched_uri:
101
+ return {
102
+ "status": "unsupported",
103
+ "error": "no_matched_uri",
104
+ "message": "Instruction could not be resolved to any registered URI",
105
+ "uri": None,
106
+ "input": {},
107
+ "domain": domain.name,
108
+ "model_used": selected_model,
109
+ "duration_ms": duration_ms,
110
+ "raw": response,
111
+ }
112
+
113
+ # AC: only selected registered URI yields success; reject invented operations
114
+ target_op = next((op for op in self.catalog if op.get("uri") == matched_uri), None)
115
+ if not target_op:
116
+ return {
117
+ "status": "unsupported",
118
+ "error": "unregistered_uri",
119
+ "message": f"Operation '{matched_uri}' is not registered in catalog",
120
+ "uri": None,
121
+ "input": {},
122
+ "domain": domain.name,
123
+ "model_used": selected_model,
124
+ "duration_ms": duration_ms,
125
+ "raw": response,
126
+ }
127
+
128
+ input_args = response.get("input", {})
129
+ schema = target_op.get("input_schema")
130
+ if schema:
131
+ try:
132
+ import jsonschema
133
+ jsonschema.validate(instance=input_args, schema=schema)
134
+ except Exception as ve:
135
+ return {
136
+ "status": "clarify",
137
+ "error": "invalid_arguments",
138
+ "message": f"Argument validation against schema failed: {ve}",
139
+ "uri": target_op["uri"],
140
+ "input": input_args,
141
+ "domain": domain.name,
142
+ "model_used": selected_model,
143
+ "duration_ms": duration_ms,
144
+ "raw": response,
145
+ }
146
+
147
+ return {
148
+ "status": "ok",
149
+ "uri": target_op["uri"],
150
+ "input": input_args,
151
+ "domain": domain.name,
152
+ "model_used": selected_model,
153
+ "duration_ms": duration_ms,
154
+ "raw": response,
155
+ }
willuri/semantic.py ADDED
@@ -0,0 +1,47 @@
1
+ """Fast local lexical matcher and candidate ranker for operations.
2
+
3
+ Uses Jaccard token overlap for lexical keyword pre-filtering across catalog operations.
4
+ Note: For semantic vector retrieval, use an embedding adapter rather than lexical overlap.
5
+ """
6
+ from __future__ import annotations
7
+ import math
8
+ import re
9
+ from typing import Any, Dict, List, Tuple
10
+
11
+
12
+ def tokenize(text: str) -> List[str]:
13
+ """Tokenize text into lowercase alphanumeric keywords."""
14
+ return [t for t in re.findall(r"\w+", text.lower()) if len(t) > 1]
15
+
16
+
17
+ def compute_similarity(tokens_a: List[str], tokens_b: List[str]) -> float:
18
+ """Compute Jaccard token similarity with term frequency weighting."""
19
+ if not tokens_a or not tokens_b:
20
+ return 0.0
21
+ set_a = set(tokens_a)
22
+ set_b = set(tokens_b)
23
+ intersection = set_a.intersection(set_b)
24
+ union = set_a.union(set_b)
25
+ return len(intersection) / len(union) if union else 0.0
26
+
27
+
28
+ class SemanticMatcher:
29
+ """Ranks and pre-filters catalog operations against natural language instructions."""
30
+
31
+ def __init__(self, catalog: List[Dict[str, Any]]):
32
+ self.catalog = catalog
33
+
34
+ def match(self, instruction: str, top_k: int = 5) -> List[Tuple[Dict[str, Any], float]]:
35
+ """Return the top_k most relevant operations with relevance scores."""
36
+ inst_tokens = tokenize(instruction)
37
+ scored: List[Tuple[Dict[str, Any], float]] = []
38
+
39
+ for op in self.catalog:
40
+ text = f"{op.get('uri', '')} {op.get('description', '')} {' '.join(op.get('aliases', {}).keys())}"
41
+ op_tokens = tokenize(text)
42
+ score = compute_similarity(inst_tokens, op_tokens)
43
+ if score > 0.0:
44
+ scored.append((op, round(score, 4)))
45
+
46
+ scored.sort(key=lambda x: x[1], reverse=True)
47
+ return scored[:top_k]
@@ -0,0 +1,57 @@
1
+ Metadata-Version: 2.4
2
+ Name: willuri
3
+ Version: 0.1.0
4
+ Summary: Natural language to specialized URI process router with domain-specific small LLMs
5
+ Author-email: Tom Sapletta <tom@sapletta.com>
6
+ License: Proprietary
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: pyyaml>=6.0
11
+ Dynamic: license-file
12
+
13
+ # willuri
14
+
15
+ Natural language to specialized process URI router powered by domain-specific lightweight LLMs.
16
+
17
+ ## Overview
18
+
19
+ In the Paxlet / Willman ecosystem, every capability is addressed by an explicit process URI (`willman://operation/...`, `dockuri://...`).
20
+ `willuri` maps natural language user instructions or tickets from the board to specialized process URIs using small, fast, domain-targeted LLMs instead of a monolithic general-purpose model.
21
+
22
+ ### Specialized Domain Models
23
+
24
+ | Domain | Process URI Scope | Primary Model | Alternative | Focus |
25
+ | :--- | :--- | :--- | :--- | :--- |
26
+ | **`code`** | `willman://operation/code.*`, `git.resolve_conflicts` | `qwen2.5-coder:3b` | `willman-nlp:qwen2.5-3b` | Code synthesis, AST edits, diffs, bug repairs |
27
+ | **`api_ops`** | `willman://operation/github.*`, `dockuri.*`, `files.*` | `granite4.1:3b` | `willman-nlp:qwen2.5-3b` | Structured API tools, JSON schema, enterprise ops |
28
+ | **`planning`** | `willman://operation/koru.*`, `planfile.*`, `schedule.*` | `llama3.2:3b` | `willman-nlp:qwen2.5-3b` | Ticket decomposition, sprint planning, DAG subtasks |
29
+ | **`text`** | `willman://operation/text.*` | `willman-nlp:qwen2.5-1.5b` | `willman-nlp:qwen2.5-3b` | String normalization, fast formatting, prose |
30
+
31
+ ## Usage
32
+
33
+ ### Python API
34
+ ```python
35
+ from willuri.router import UriRouter
36
+
37
+ router = UriRouter()
38
+ router.add_operation("willman://operation/text.normalize/v1", "Normalize whitespace")
39
+ router.add_operation("willman://operation/git.resolve_conflicts/v1", "Resolve git merge conflicts")
40
+
41
+ result = router.route("Rozwiąż konflikty git w repozytorium")
42
+ print(result["uri"])
43
+ # => "willman://operation/git.resolve_conflicts/v1"
44
+ print(result["model_used"])
45
+ # => "qwen2.5-coder:3b"
46
+ ```
47
+
48
+ ### CLI
49
+ ```bash
50
+ willuri classify "Popraw błąd w funkcji liczącej sumę"
51
+ willuri route "Znormalizuj odstępy w tekście 'Ala ma kota'"
52
+ ```
53
+
54
+ ## Verification
55
+ ```bash
56
+ python3 -m unittest discover -s tests -v
57
+ ```
@@ -0,0 +1,15 @@
1
+ willuri/__init__.py,sha256=2UduPG-PNE4NwN3eFSYlt6xP20eKbo7BZg2K9LoX9rs,913
2
+ willuri/cli.py,sha256=DMGPCANefZmqcPisVSbmbN0MhKAVbkgXg7k9YMPwQg8,2137
3
+ willuri/client.py,sha256=dhN-6GXKhhjIA1EtfPb0-dnTcCAGcdx_ha6C2UCZ9y4,1170
4
+ willuri/domains.py,sha256=zjXs4KuT2pGhovBLrMFsruDRTkdL6GBIFCJAIuL-9Bs,3632
5
+ willuri/dsl.py,sha256=Mx3OZc1gM5DvSrZzrgmaRFVAo7ntkKM2BFzROOTf0BE,2123
6
+ willuri/models.py,sha256=0K4WiQQGNT-HDac3gFQBQMVdAx_1ndGKUOIcrjSMfro,3269
7
+ willuri/prompts.py,sha256=HkYtMcp1BzTxfGCp2gf9fGh4Pr_1iYvKKv5roifJCyM,1858
8
+ willuri/router.py,sha256=tens17O9Mjlnqkj3jA1ZkYB2y_Rej4rC5Oblc6pa92M,5718
9
+ willuri/semantic.py,sha256=E0QHdKia_w1dcT5ANUIgKGUjN9m13-v8LlfSFxlzYSA,1805
10
+ willuri-0.1.0.dist-info/licenses/LICENSE,sha256=rtJN2PQxbee7jzTmeZ63miF9k-OVgiojG8uodj7k-zE,6762
11
+ willuri-0.1.0.dist-info/METADATA,sha256=QqF4_hvsUuiXfBBRUVMh8OVTlbK3VzJk7fYLHQ8OgaE,2289
12
+ willuri-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
13
+ willuri-0.1.0.dist-info/entry_points.txt,sha256=FYhMOxK1R306eoXj_wODmKM_ASTGqJIaGgQU6_sC-BE,45
14
+ willuri-0.1.0.dist-info/top_level.txt,sha256=fIXohf-YQe-BnCPfp2gGsOOLoyWqR24kHGaLFBJ5RZ4,8
15
+ willuri-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ willuri = willuri.cli:main
@@ -0,0 +1,110 @@
1
+ PROPRIETARY AND CONFIDENTIAL SOURCE CODE LICENSE
2
+ STRICT ANTI-AI / ANTI-MINING RESERVATION OF RIGHTS
3
+
4
+ Copyright (c) 2026 Tomasz Sapletta Prototypowanie.pl NIP: 5881918662, REGON: 220665410. All rights reserved.
5
+ Właścicielem oprogramowania jest Tomasz Sapletta Prototypowanie.pl NIP: 5881918662, REGON: 220665410.
6
+
7
+ ================================================================================
8
+ IMPORTANT NOTICE - PROPRIETARY AND CONFIDENTIAL
9
+ ================================================================================
10
+
11
+ This software, source code, documentation, specifications, schemas, models,
12
+ algorithms, and associated digital assets ("the Software") are the confidential
13
+ and proprietary intellectual property of Tomasz Sapletta Prototypowanie.pl
14
+ (NIP: 5881918662, REGON: 220665410) and its licensors.
15
+
16
+ The Software is NOT licensed under any open source, free software, or
17
+ source-available license (such as MIT, Apache 2.0, BSD, GPL, AGPL, BSL, or SSPL).
18
+ All rights not expressly granted under a separate, signed written commercial
19
+ agreement executed by Tomasz Sapletta Prototypowanie.pl are strictly reserved.
20
+
21
+ ================================================================================
22
+ SECTION 1. STRICT PROHIBITION ON ARTIFICIAL INTELLIGENCE & MACHINE LEARNING INGESTION
23
+ ================================================================================
24
+
25
+ Under no circumstances may any portion of the Software, whether in human-readable
26
+ source code, compiled binary, abstract syntax tree (AST), intermediate representation,
27
+ documentation, ticket history, or commit metadata, be:
28
+
29
+ 1.1. Ingested, scraped, crawled, harvested, indexed, tokenized, cached, or stored
30
+ for the purpose of training, pre-training, fine-tuning, evaluating, or weighting
31
+ any artificial intelligence system, machine learning model, large language
32
+ model (LLM), code-generation model, neural network, diffusion model, or
33
+ multimodal model.
34
+
35
+ 1.2. Used by, or fed into the training datasets, prompt contexts, retrieval-augmented
36
+ generation (RAG) stores, vector databases, or telemetry caches of any commercial
37
+ or non-commercial AI system, including but not limited to:
38
+ - GitHub Copilot / Microsoft Copilot (Microsoft Corporation)
39
+ - OpenAI ChatGPT, Codex, GPT-4, GPT-5, o-series, and related APIs (OpenAI Inc.)
40
+ - Google Gemini, PaLM, Codey, Vertex AI, and DeepMind systems (Google LLC / Alphabet)
41
+ - Anthropic Claude, Constitutional AI, and related models (Anthropic PBC)
42
+ - Meta LLaMA, Code Llama, and related foundation models (Meta Platforms, Inc.)
43
+ - Amazon CodeWhisperer, Bedrock, and Titan models (Amazon.com, Inc.)
44
+ - Mistral AI, Cohere, Cursor AI, Tabnine, Replit Ghostwriter, or any automated
45
+ code-synthesis engine or AI coding assistant.
46
+
47
+ 1.3. Processed or analyzed by any automated tool, bot, crawler, or scraper to generate
48
+ synthetic code, code suggestions, autocomplete suggestions, embeddings, or
49
+ derivative software models.
50
+
51
+ ================================================================================
52
+ SECTION 2. STATUTORY RESERVATION OF RIGHTS (EU DIRECTIVE 2019/790)
53
+ ================================================================================
54
+
55
+ 2.1. In accordance with Article 4(3) of Directive (EU) 2019/790 of the European
56
+ Parliament and of the Council on copyright and related rights in the Digital
57
+ Single Market, the rights holder hereby EXPRESSLY RESERVES ALL RIGHTS
58
+ regarding Text and Data Mining (TDM) for this Software, its source repositories,
59
+ documentation, schemas, and all associated materials.
60
+
61
+ 2.2. Any extraction, automated reading, indexing, or reproduction of the Software
62
+ for text and data mining purposes within the meaning of Directive (EU) 2019/790
63
+ is strictly prohibited without prior written authorization from the rights holder.
64
+
65
+ 2.3. This reservation applies worldwide and is declared in machine-readable form
66
+ within the repository metadata, headers, and this License specification.
67
+
68
+ ================================================================================
69
+ SECTION 3. HOSTING & PLATFORM PREEMPTION CLAUSE
70
+ ================================================================================
71
+
72
+ 3.1. The presence of this source code or repository on any third-party hosting service,
73
+ code forge, version control platform, or cloud provider (including but not
74
+ limited to GitHub, GitLab, Bitbucket, or AWS) does NOT constitute a waiver,
75
+ license grant, or implied consent to the hosting provider, its parent,
76
+ subsidiaries, or third parties to:
77
+ (a) use the Software for machine learning training, telemetry harvesting, or
78
+ model improvement;
79
+ (b) distribute or reproduce the Software to other users or automated agents;
80
+ (c) index the Software in any public search or retrieval system.
81
+
82
+ 3.2. Any terms of service, platform agreements, or automated licenses purporting
83
+ to grant the platform operator an implied, royalty-free, or transferable
84
+ license to utilize hosted code for artificial intelligence development, model
85
+ training, or feature synthesis are hereby explicitly rejected, preempted,
86
+ and declared null and void with respect to this Software.
87
+
88
+ ================================================================================
89
+ SECTION 4. PROHIBITED USES & RESTRICTIONS
90
+ ================================================================================
91
+
92
+ Except as expressly authorized under a valid, executed commercial enterprise license
93
+ with Tomasz Sapletta Prototypowanie.pl, you may NOT:
94
+ - Copy, modify, adapt, translate, reverse engineer, decompile, or disassemble the Software.
95
+ - Distribute, sublicense, rent, lease, lend, sell, or publicly display the Software.
96
+ - Use the Software to provide competitive time-sharing, SaaS, cloud twin, or virtualization services.
97
+ - Remove, alter, or obscure any proprietary notices, copyright labels, or license files.
98
+
99
+ ================================================================================
100
+ SECTION 5. ENFORCEMENT & REMEDIES
101
+ ================================================================================
102
+
103
+ Any breach of this License, including unauthorized scraping, tokenization, or ingestion
104
+ into an artificial intelligence system, constitutes willful copyright infringement,
105
+ unauthorized access, and breach of contract. Tomasz Sapletta Prototypowanie.pl reserves
106
+ the right to seek immediate injunctive relief, statutory damages, compensation for commercial
107
+ harm, and mandatory destruction of all derivative model weights, embeddings, and training
108
+ caches containing representations of the Software.
109
+
110
+ For licensing inquiries and commercial authorization, contact: tom@prototypowanie.pl / legal@clonerd.com
@@ -0,0 +1 @@
1
+ willuri