narrow-mcp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- narrow_mcp-0.1.0/LICENSE +21 -0
- narrow_mcp-0.1.0/PKG-INFO +126 -0
- narrow_mcp-0.1.0/README.md +98 -0
- narrow_mcp-0.1.0/pyproject.toml +50 -0
- narrow_mcp-0.1.0/setup.cfg +4 -0
- narrow_mcp-0.1.0/src/narrow_mcp/__init__.py +1 -0
- narrow_mcp-0.1.0/src/narrow_mcp/budget.py +39 -0
- narrow_mcp-0.1.0/src/narrow_mcp/config.py +78 -0
- narrow_mcp-0.1.0/src/narrow_mcp/file_type_detector.py +113 -0
- narrow_mcp-0.1.0/src/narrow_mcp/list_free_models.py +74 -0
- narrow_mcp-0.1.0/src/narrow_mcp/mcp_handler.py +56 -0
- narrow_mcp-0.1.0/src/narrow_mcp/models.py +107 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/__init__.py +10 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/csv_type.py +201 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/html_type.py +220 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/json_type.py +303 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/keyword_hints.toml +35 -0
- narrow_mcp-0.1.0/src/narrow_mcp/narrowing/log_text.py +172 -0
- narrow_mcp-0.1.0/src/narrow_mcp/orchestrator.py +112 -0
- narrow_mcp-0.1.0/src/narrow_mcp/selector_client.py +252 -0
- narrow_mcp-0.1.0/src/narrow_mcp/verifier.py +66 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/PKG-INFO +126 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/SOURCES.txt +25 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/dependency_links.txt +1 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/entry_points.txt +3 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/requires.txt +8 -0
- narrow_mcp-0.1.0/src/narrow_mcp.egg-info/top_level.txt +1 -0
narrow_mcp-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Manik Prakash
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: narrow-mcp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP server that narrows large low-density files (logs, CSV, JSON, HTML) to the verbatim spans relevant to an intent, verified against source.
|
|
5
|
+
Author: Manik Prakash
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/manik-prakash/narrow-mcp
|
|
8
|
+
Project-URL: Issues, https://github.com/manik-prakash/narrow-mcp/issues
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Topic :: Software Development :: Build Tools
|
|
17
|
+
Requires-Python: >=3.11
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: mcp[cli]>=2.0.0
|
|
21
|
+
Requires-Dist: anthropic>=0.40.0
|
|
22
|
+
Requires-Dist: openai>=1.0.0
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: pytest>=8.0.0; extra == "dev"
|
|
25
|
+
Requires-Dist: pytest-asyncio>=0.24.0; extra == "dev"
|
|
26
|
+
Requires-Dist: pyyaml>=6.0; extra == "dev"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# narrow-mcp
|
|
30
|
+
|
|
31
|
+
<!-- mcp-name: io.github.manik-prakash/narrow-mcp -->
|
|
32
|
+
|
|
33
|
+
An MCP server that narrows large, low-density non-source files (logs, test
|
|
34
|
+
output, CSV, JSON, HTML) down to the verbatim spans relevant to a stated
|
|
35
|
+
intent -- verified against the original file, never a summary.
|
|
36
|
+
|
|
37
|
+
## Why
|
|
38
|
+
|
|
39
|
+
Coding agents burn context reading large low-density files. A 40k-token log
|
|
40
|
+
might hold a few hundred tokens of signal. This tool:
|
|
41
|
+
|
|
42
|
+
1. Runs deterministic narrowing first (grep-style search, structural
|
|
43
|
+
parsing per file type, sampling) -- free, fast, zero LLM cost.
|
|
44
|
+
2. If that alone resolves the query with confidence (e.g. an exact CSV
|
|
45
|
+
"null column" query), returns it directly. No LLM call at all.
|
|
46
|
+
3. Otherwise passes only the narrowed candidates (never the raw file) to a
|
|
47
|
+
cheap, fast selector model that returns line ranges and a one-line
|
|
48
|
+
reason -- never prose.
|
|
49
|
+
4. Re-reads the chosen line ranges from the **original file on disk** and
|
|
50
|
+
returns that verbatim text. The selector's own words are never trusted
|
|
51
|
+
or returned -- only its line-number coordinates, which get verified.
|
|
52
|
+
5. If the selector fails, times out, or returns something invalid, falls
|
|
53
|
+
back to the deterministic candidate set rather than failing outright.
|
|
54
|
+
|
|
55
|
+
Non-goals: source code retrieval (use LSP/tree-sitter/ast-grep), prose
|
|
56
|
+
summarization, local/self-hosted models.
|
|
57
|
+
|
|
58
|
+
## Status
|
|
59
|
+
|
|
60
|
+
v1, single file per call. Four file types: log/build-output, CSV, JSON
|
|
61
|
+
(single document or JSONL), HTML.
|
|
62
|
+
|
|
63
|
+
## Development
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
pip install -e ".[dev]"
|
|
67
|
+
pytest
|
|
68
|
+
python eval/run_eval.py # mocked selector, free
|
|
69
|
+
python eval/run_eval.py --live # real selector call, needs an API key (see Configuration)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Configuration
|
|
73
|
+
|
|
74
|
+
**You only need to set one API key.** The provider is auto-detected from
|
|
75
|
+
whichever key is present -- no separate provider/model config required:
|
|
76
|
+
|
|
77
|
+
| If you set... | Provider used | Default model |
|
|
78
|
+
|----------------------|----------------|-------------------------------------------------|
|
|
79
|
+
| `ANTHROPIC_API_KEY` | `anthropic` | `claude-haiku-4-5` |
|
|
80
|
+
| `OPENAI_API_KEY` | `openai` | `gpt-5-nano` |
|
|
81
|
+
| `OPENROUTER_API_KEY` | `openrouter` | `openrouter/free` (see below) |
|
|
82
|
+
| *(none)* | `anthropic` | `claude-haiku-4-5` (calls just always fall back to the deterministic path) |
|
|
83
|
+
|
|
84
|
+
If more than one key is set, priority is Anthropic > OpenAI > OpenRouter.
|
|
85
|
+
Override anything explicitly with the env vars below.
|
|
86
|
+
|
|
87
|
+
### Using OpenRouter's free models
|
|
88
|
+
|
|
89
|
+
OpenRouter still requires its own API key even for $0-cost models -- set
|
|
90
|
+
`OPENROUTER_API_KEY` and you're done, no other config needed. It defaults to
|
|
91
|
+
**`openrouter/free`**, a meta-router that auto-picks among whichever
|
|
92
|
+
tool-calling-capable models are currently free, so it never goes stale the
|
|
93
|
+
way hardcoding one specific `:free` model name would.
|
|
94
|
+
|
|
95
|
+
To see the current free-model roster live (it rotates) and pick a specific
|
|
96
|
+
one instead of the meta-router:
|
|
97
|
+
|
|
98
|
+
```
|
|
99
|
+
narrow-mcp-list-free-models
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Then set `NARROW_MCP_SELECTOR_MODEL=<id>` to whichever one you want.
|
|
103
|
+
|
|
104
|
+
OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, or 1000/day
|
|
105
|
+
once the account has $10+ lifetime spend) -- fine for interactive use, worth
|
|
106
|
+
knowing about for batch runs.
|
|
107
|
+
|
|
108
|
+
### All environment variables
|
|
109
|
+
|
|
110
|
+
- `NARROW_MCP_SELECTOR_PROVIDER` -- `anthropic` | `openai` | `openrouter`.
|
|
111
|
+
Overrides auto-detection.
|
|
112
|
+
- `NARROW_MCP_SELECTOR_MODEL` -- overrides the provider's default model.
|
|
113
|
+
- `NARROW_MCP_SELECTOR_API_KEY_ENV` -- overrides which env var holds the key.
|
|
114
|
+
- `NARROW_MCP_SELECTOR_TIMEOUT_S` (default `3.0`)
|
|
115
|
+
- `NARROW_MCP_MAX_CANDIDATE_CHARS`, `NARROW_MCP_MAX_CANDIDATES`,
|
|
116
|
+
`NARROW_MCP_CONTEXT_LINES`, `NARROW_MCP_RIPGREP_PATH`,
|
|
117
|
+
`NARROW_MCP_MAX_JSON_BYTES`
|
|
118
|
+
|
|
119
|
+
## Registering with Claude Code
|
|
120
|
+
|
|
121
|
+
```
|
|
122
|
+
claude mcp add narrow-mcp -- uvx narrow-mcp
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Verify current `claude mcp add` flag syntax against `claude mcp add --help`
|
|
126
|
+
before relying on the above -- CLI flags change across releases.
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# narrow-mcp
|
|
2
|
+
|
|
3
|
+
<!-- mcp-name: io.github.manik-prakash/narrow-mcp -->
|
|
4
|
+
|
|
5
|
+
An MCP server that narrows large, low-density non-source files (logs, test
|
|
6
|
+
output, CSV, JSON, HTML) down to the verbatim spans relevant to a stated
|
|
7
|
+
intent -- verified against the original file, never a summary.
|
|
8
|
+
|
|
9
|
+
## Why
|
|
10
|
+
|
|
11
|
+
Coding agents burn context reading large low-density files. A 40k-token log
|
|
12
|
+
might hold a few hundred tokens of signal. This tool:
|
|
13
|
+
|
|
14
|
+
1. Runs deterministic narrowing first (grep-style search, structural
|
|
15
|
+
parsing per file type, sampling) -- free, fast, zero LLM cost.
|
|
16
|
+
2. If that alone resolves the query with confidence (e.g. an exact CSV
|
|
17
|
+
"null column" query), returns it directly. No LLM call at all.
|
|
18
|
+
3. Otherwise passes only the narrowed candidates (never the raw file) to a
|
|
19
|
+
cheap, fast selector model that returns line ranges and a one-line
|
|
20
|
+
reason -- never prose.
|
|
21
|
+
4. Re-reads the chosen line ranges from the **original file on disk** and
|
|
22
|
+
returns that verbatim text. The selector's own words are never trusted
|
|
23
|
+
or returned -- only its line-number coordinates, which get verified.
|
|
24
|
+
5. If the selector fails, times out, or returns something invalid, falls
|
|
25
|
+
back to the deterministic candidate set rather than failing outright.
|
|
26
|
+
|
|
27
|
+
Non-goals: source code retrieval (use LSP/tree-sitter/ast-grep), prose
|
|
28
|
+
summarization, local/self-hosted models.
|
|
29
|
+
|
|
30
|
+
## Status
|
|
31
|
+
|
|
32
|
+
v1, single file per call. Four file types: log/build-output, CSV, JSON
|
|
33
|
+
(single document or JSONL), HTML.
|
|
34
|
+
|
|
35
|
+
## Development
|
|
36
|
+
|
|
37
|
+
```
|
|
38
|
+
pip install -e ".[dev]"
|
|
39
|
+
pytest
|
|
40
|
+
python eval/run_eval.py # mocked selector, free
|
|
41
|
+
python eval/run_eval.py --live # real selector call, needs an API key (see Configuration)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Configuration
|
|
45
|
+
|
|
46
|
+
**You only need to set one API key.** The provider is auto-detected from
|
|
47
|
+
whichever key is present -- no separate provider/model config required:
|
|
48
|
+
|
|
49
|
+
| If you set... | Provider used | Default model |
|
|
50
|
+
|----------------------|----------------|-------------------------------------------------|
|
|
51
|
+
| `ANTHROPIC_API_KEY` | `anthropic` | `claude-haiku-4-5` |
|
|
52
|
+
| `OPENAI_API_KEY` | `openai` | `gpt-5-nano` |
|
|
53
|
+
| `OPENROUTER_API_KEY` | `openrouter` | `openrouter/free` (see below) |
|
|
54
|
+
| *(none)* | `anthropic` | `claude-haiku-4-5` (calls just always fall back to the deterministic path) |
|
|
55
|
+
|
|
56
|
+
If more than one key is set, priority is Anthropic > OpenAI > OpenRouter.
|
|
57
|
+
Override anything explicitly with the env vars below.
|
|
58
|
+
|
|
59
|
+
### Using OpenRouter's free models
|
|
60
|
+
|
|
61
|
+
OpenRouter still requires its own API key even for $0-cost models -- set
|
|
62
|
+
`OPENROUTER_API_KEY` and you're done, no other config needed. It defaults to
|
|
63
|
+
**`openrouter/free`**, a meta-router that auto-picks among whichever
|
|
64
|
+
tool-calling-capable models are currently free, so it never goes stale the
|
|
65
|
+
way hardcoding one specific `:free` model name would.
|
|
66
|
+
|
|
67
|
+
To see the current free-model roster live (it rotates) and pick a specific
|
|
68
|
+
one instead of the meta-router:
|
|
69
|
+
|
|
70
|
+
```
|
|
71
|
+
narrow-mcp-list-free-models
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Then set `NARROW_MCP_SELECTOR_MODEL=<id>` to whichever one you want.
|
|
75
|
+
|
|
76
|
+
OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, or 1000/day
|
|
77
|
+
once the account has $10+ lifetime spend) -- fine for interactive use, worth
|
|
78
|
+
knowing about for batch runs.
|
|
79
|
+
|
|
80
|
+
### All environment variables
|
|
81
|
+
|
|
82
|
+
- `NARROW_MCP_SELECTOR_PROVIDER` -- `anthropic` | `openai` | `openrouter`.
|
|
83
|
+
Overrides auto-detection.
|
|
84
|
+
- `NARROW_MCP_SELECTOR_MODEL` -- overrides the provider's default model.
|
|
85
|
+
- `NARROW_MCP_SELECTOR_API_KEY_ENV` -- overrides which env var holds the key.
|
|
86
|
+
- `NARROW_MCP_SELECTOR_TIMEOUT_S` (default `3.0`)
|
|
87
|
+
- `NARROW_MCP_MAX_CANDIDATE_CHARS`, `NARROW_MCP_MAX_CANDIDATES`,
|
|
88
|
+
`NARROW_MCP_CONTEXT_LINES`, `NARROW_MCP_RIPGREP_PATH`,
|
|
89
|
+
`NARROW_MCP_MAX_JSON_BYTES`
|
|
90
|
+
|
|
91
|
+
## Registering with Claude Code
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
claude mcp add narrow-mcp -- uvx narrow-mcp
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Verify current `claude mcp add` flag syntax against `claude mcp add --help`
|
|
98
|
+
before relying on the above -- CLI flags change across releases.
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "narrow-mcp"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "MCP server that narrows large low-density files (logs, CSV, JSON, HTML) to the verbatim spans relevant to an intent, verified against source."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
authors = [{ name = "Manik Prakash" }]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 4 - Beta",
|
|
12
|
+
"Intended Audience :: Developers",
|
|
13
|
+
"Programming Language :: Python :: 3",
|
|
14
|
+
"Programming Language :: Python :: 3.11",
|
|
15
|
+
"Programming Language :: Python :: 3.12",
|
|
16
|
+
"Programming Language :: Python :: 3.13",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
"Topic :: Software Development :: Build Tools",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"mcp[cli]>=2.0.0",
|
|
22
|
+
"anthropic>=0.40.0",
|
|
23
|
+
"openai>=1.0.0",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.urls]
|
|
27
|
+
Repository = "https://github.com/manik-prakash/narrow-mcp"
|
|
28
|
+
Issues = "https://github.com/manik-prakash/narrow-mcp/issues"
|
|
29
|
+
|
|
30
|
+
[project.scripts]
|
|
31
|
+
narrow-mcp = "narrow_mcp.mcp_handler:main"
|
|
32
|
+
narrow-mcp-list-free-models = "narrow_mcp.list_free_models:main"
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
dev = ["pytest>=8.0.0", "pytest-asyncio>=0.24.0", "pyyaml>=6.0"]
|
|
36
|
+
|
|
37
|
+
[build-system]
|
|
38
|
+
requires = ["setuptools>=77.0.0"]
|
|
39
|
+
build-backend = "setuptools.build_meta"
|
|
40
|
+
|
|
41
|
+
[tool.setuptools.packages.find]
|
|
42
|
+
where = ["src"]
|
|
43
|
+
|
|
44
|
+
[tool.setuptools.package-data]
|
|
45
|
+
narrow_mcp = ["narrowing/keyword_hints.toml"]
|
|
46
|
+
|
|
47
|
+
[tool.pytest.ini_options]
|
|
48
|
+
pythonpath = ["src"]
|
|
49
|
+
testpaths = ["tests"]
|
|
50
|
+
asyncio_mode = "auto"
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""narrow_mcp: narrows large low-density files to the spans relevant to an intent."""
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Candidate-budget enforcement. Standalone so policy can be tuned without touching narrowing logic."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .config import Budget
|
|
6
|
+
from .models import Candidate
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def enforce(candidates: list[Candidate], budget: Budget) -> tuple[list[Candidate], int]:
|
|
10
|
+
"""Truncate/prioritize candidates to fit the budget.
|
|
11
|
+
|
|
12
|
+
Returns (kept_candidates, truncated_count). Always keeps at least one
|
|
13
|
+
candidate if any were given. Prioritizes by score (highest first), then
|
|
14
|
+
by document order for ties.
|
|
15
|
+
"""
|
|
16
|
+
if not candidates:
|
|
17
|
+
return [], 0
|
|
18
|
+
|
|
19
|
+
ordered = sorted(
|
|
20
|
+
enumerate(candidates), key=lambda pair: (-pair[1].score, pair[0])
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
kept: list[Candidate] = []
|
|
24
|
+
total_chars = 0
|
|
25
|
+
for _, candidate in ordered:
|
|
26
|
+
preview_len = len(candidate.preview_text)
|
|
27
|
+
if kept and (
|
|
28
|
+
len(kept) >= budget.max_candidates
|
|
29
|
+
or total_chars + preview_len > budget.max_candidate_chars
|
|
30
|
+
):
|
|
31
|
+
continue
|
|
32
|
+
kept.append(candidate)
|
|
33
|
+
total_chars += preview_len
|
|
34
|
+
|
|
35
|
+
# Restore original document order for readability in the prompt.
|
|
36
|
+
kept_ids = {id(c) for c in kept}
|
|
37
|
+
kept_in_order = [c for c in candidates if id(c) in kept_ids]
|
|
38
|
+
truncated_count = len(candidates) - len(kept_in_order)
|
|
39
|
+
return kept_in_order, truncated_count
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""All tunables live here. Loaded from environment variables with sane defaults."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class Budget:
|
|
11
|
+
max_candidate_chars: int = 8000
|
|
12
|
+
max_candidates: int = 30
|
|
13
|
+
max_context_lines_per_hit: int = 3
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class SelectorConfig:
|
|
18
|
+
provider: str = "anthropic"
|
|
19
|
+
model: str = "claude-haiku-4-5"
|
|
20
|
+
api_key_env_var: str = "ANTHROPIC_API_KEY"
|
|
21
|
+
timeout_s: float = 3.0
|
|
22
|
+
max_retries: int = 0
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
_PROVIDER_DEFAULTS: dict[str, dict[str, str]] = {
|
|
26
|
+
"anthropic": {"model": "claude-haiku-4-5", "api_key_env_var": "ANTHROPIC_API_KEY"},
|
|
27
|
+
"openai": {"model": "gpt-5-nano", "api_key_env_var": "OPENAI_API_KEY"},
|
|
28
|
+
# openrouter/free is a stable meta-router that auto-picks among currently
|
|
29
|
+
# free, tool-calling-capable models -- never goes stale the way a specific
|
|
30
|
+
# ":free" model id would as OpenRouter's free roster rotates.
|
|
31
|
+
"openrouter": {"model": "openrouter/free", "api_key_env_var": "OPENROUTER_API_KEY"},
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _autodetect_provider() -> str:
|
|
36
|
+
"""Only used when NARROW_MCP_SELECTOR_PROVIDER is not explicitly set.
|
|
37
|
+
Priority: a directly-paid-for native provider before OpenRouter (the
|
|
38
|
+
$0-cost fallback, not the preferred option if the user already has a
|
|
39
|
+
native key). No key at all -> today's unchanged default.
|
|
40
|
+
"""
|
|
41
|
+
if os.environ.get("ANTHROPIC_API_KEY"):
|
|
42
|
+
return "anthropic"
|
|
43
|
+
if os.environ.get("OPENAI_API_KEY"):
|
|
44
|
+
return "openai"
|
|
45
|
+
if os.environ.get("OPENROUTER_API_KEY"):
|
|
46
|
+
return "openrouter"
|
|
47
|
+
return "anthropic"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class Config:
|
|
52
|
+
budget: Budget
|
|
53
|
+
selector: SelectorConfig
|
|
54
|
+
ripgrep_path: str | None = None
|
|
55
|
+
max_json_bytes: int = 20_000_000
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def load(cls) -> "Config":
|
|
59
|
+
budget = Budget(
|
|
60
|
+
max_candidate_chars=int(os.environ.get("NARROW_MCP_MAX_CANDIDATE_CHARS", 8000)),
|
|
61
|
+
max_candidates=int(os.environ.get("NARROW_MCP_MAX_CANDIDATES", 30)),
|
|
62
|
+
max_context_lines_per_hit=int(os.environ.get("NARROW_MCP_CONTEXT_LINES", 3)),
|
|
63
|
+
)
|
|
64
|
+
provider = os.environ.get("NARROW_MCP_SELECTOR_PROVIDER") or _autodetect_provider()
|
|
65
|
+
defaults = _PROVIDER_DEFAULTS.get(provider, _PROVIDER_DEFAULTS["anthropic"])
|
|
66
|
+
selector = SelectorConfig(
|
|
67
|
+
provider=provider,
|
|
68
|
+
model=os.environ.get("NARROW_MCP_SELECTOR_MODEL", defaults["model"]),
|
|
69
|
+
api_key_env_var=os.environ.get("NARROW_MCP_SELECTOR_API_KEY_ENV", defaults["api_key_env_var"]),
|
|
70
|
+
timeout_s=float(os.environ.get("NARROW_MCP_SELECTOR_TIMEOUT_S", 3.0)),
|
|
71
|
+
max_retries=int(os.environ.get("NARROW_MCP_SELECTOR_MAX_RETRIES", 0)),
|
|
72
|
+
)
|
|
73
|
+
return cls(
|
|
74
|
+
budget=budget,
|
|
75
|
+
selector=selector,
|
|
76
|
+
ripgrep_path=os.environ.get("NARROW_MCP_RIPGREP_PATH"),
|
|
77
|
+
max_json_bytes=int(os.environ.get("NARROW_MCP_MAX_JSON_BYTES", 20_000_000)),
|
|
78
|
+
)
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Decide which narrowing strategy applies to a file, and refuse source/binary files.
|
|
2
|
+
|
|
3
|
+
This is the single place the "no source-code retrieval" non-goal is enforced.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import csv
|
|
9
|
+
import io
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from .models import FileTypeDecision
|
|
14
|
+
|
|
15
|
+
SOURCE_EXTENSIONS = {
|
|
16
|
+
".py", ".pyi", ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs",
|
|
17
|
+
".go", ".rs", ".java", ".kt", ".kts", ".c", ".h", ".cpp", ".hpp", ".cc",
|
|
18
|
+
".cs", ".rb", ".php", ".swift", ".scala", ".m", ".mm", ".sh", ".ps1",
|
|
19
|
+
".sql", ".lua", ".r", ".pl", ".ex", ".exs", ".clj", ".hs", ".dart",
|
|
20
|
+
".vue", ".svelte",
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
CSV_EXTENSIONS = {".csv", ".tsv"}
|
|
24
|
+
JSON_EXTENSIONS = {".json", ".jsonl", ".ndjson"}
|
|
25
|
+
HTML_EXTENSIONS = {".html", ".htm", ".xhtml"}
|
|
26
|
+
LOG_EXTENSIONS = {".log", ".txt", ".out", ".err"}
|
|
27
|
+
|
|
28
|
+
SNIFF_BYTES = 8192
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _is_binary(sample: bytes) -> bool:
|
|
32
|
+
return b"\x00" in sample
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _sniff_json(sample: bytes) -> bool:
|
|
36
|
+
text = sample.decode("utf-8", errors="ignore").strip()
|
|
37
|
+
if not text:
|
|
38
|
+
return False
|
|
39
|
+
first = text[0]
|
|
40
|
+
if first not in "{[":
|
|
41
|
+
# Could still be JSONL where first line starts with { or [.
|
|
42
|
+
first_line = text.splitlines()[0].strip() if text.splitlines() else ""
|
|
43
|
+
if not first_line or first_line[0] not in "{[":
|
|
44
|
+
return False
|
|
45
|
+
try:
|
|
46
|
+
# Try parsing just the first line as a JSON value (JSONL check) or
|
|
47
|
+
# the whole sample as a prefix of a larger JSON document.
|
|
48
|
+
first_line = text.splitlines()[0]
|
|
49
|
+
json.loads(first_line)
|
|
50
|
+
return True
|
|
51
|
+
except (json.JSONDecodeError, ValueError):
|
|
52
|
+
# Might be a large single JSON document where the first line alone
|
|
53
|
+
# doesn't parse (e.g. pretty-printed). Fall back to bracket check.
|
|
54
|
+
return first in "{["
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _sniff_csv(sample: bytes) -> bool:
|
|
58
|
+
text = sample.decode("utf-8", errors="ignore")
|
|
59
|
+
if not text.strip():
|
|
60
|
+
return False
|
|
61
|
+
try:
|
|
62
|
+
csv.Sniffer().sniff(text, delimiters=",;\t|")
|
|
63
|
+
return True
|
|
64
|
+
except csv.Error:
|
|
65
|
+
return False
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _sniff_html(sample: bytes) -> bool:
|
|
69
|
+
text = sample.decode("utf-8", errors="ignore").lstrip()
|
|
70
|
+
lowered = text.lower()
|
|
71
|
+
return lowered.startswith("<!doctype html") or lowered.startswith("<html") or "<body" in lowered[:2000]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def detect(path: Path) -> FileTypeDecision:
|
|
75
|
+
if not path.exists() or not path.is_file():
|
|
76
|
+
return FileTypeDecision(kind="unknown", confidence=0.0, reason="path does not exist or is not a file")
|
|
77
|
+
|
|
78
|
+
ext = path.suffix.lower()
|
|
79
|
+
|
|
80
|
+
if ext in SOURCE_EXTENSIONS:
|
|
81
|
+
return FileTypeDecision(
|
|
82
|
+
kind="source_refused",
|
|
83
|
+
confidence=1.0,
|
|
84
|
+
reason="source file — use LSP/tree-sitter/ast-grep tooling instead",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
try:
|
|
88
|
+
with open(path, "rb") as f:
|
|
89
|
+
sample = f.read(SNIFF_BYTES)
|
|
90
|
+
except OSError as exc:
|
|
91
|
+
return FileTypeDecision(kind="unknown", confidence=0.0, reason=f"could not read file: {exc}")
|
|
92
|
+
|
|
93
|
+
if _is_binary(sample):
|
|
94
|
+
return FileTypeDecision(kind="binary_refused", confidence=1.0, reason="binary file not supported")
|
|
95
|
+
|
|
96
|
+
if ext in JSON_EXTENSIONS:
|
|
97
|
+
return FileTypeDecision(kind="json", confidence=1.0, reason=f"extension {ext}")
|
|
98
|
+
if ext in CSV_EXTENSIONS:
|
|
99
|
+
return FileTypeDecision(kind="csv", confidence=1.0, reason=f"extension {ext}")
|
|
100
|
+
if ext in HTML_EXTENSIONS:
|
|
101
|
+
return FileTypeDecision(kind="html", confidence=1.0, reason=f"extension {ext}")
|
|
102
|
+
if ext in LOG_EXTENSIONS:
|
|
103
|
+
return FileTypeDecision(kind="log", confidence=0.9, reason=f"extension {ext}")
|
|
104
|
+
|
|
105
|
+
# No recognized extension — sniff content.
|
|
106
|
+
if _sniff_json(sample):
|
|
107
|
+
return FileTypeDecision(kind="json", confidence=0.7, reason="content sniff: JSON-like")
|
|
108
|
+
if _sniff_html(sample):
|
|
109
|
+
return FileTypeDecision(kind="html", confidence=0.7, reason="content sniff: HTML-like")
|
|
110
|
+
if _sniff_csv(sample):
|
|
111
|
+
return FileTypeDecision(kind="csv", confidence=0.6, reason="content sniff: CSV-like")
|
|
112
|
+
|
|
113
|
+
return FileTypeDecision(kind="log", confidence=0.4, reason="no structured format detected, treating as text/log")
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Standalone setup-time helper -- NOT an MCP tool, not wired into the
|
|
2
|
+
server. Prints OpenRouter's currently-free, tool-calling-capable models,
|
|
3
|
+
fetched live (the free roster rotates, so this is never hardcoded).
|
|
4
|
+
|
|
5
|
+
Usage:
|
|
6
|
+
narrow-mcp-list-free-models
|
|
7
|
+
python -m narrow_mcp.list_free_models
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import sys
|
|
15
|
+
import urllib.error
|
|
16
|
+
import urllib.request
|
|
17
|
+
|
|
18
|
+
MODELS_URL = "https://openrouter.ai/api/v1/models?supported_parameters=tools"
|
|
19
|
+
RECOMMENDED_MODEL = "openrouter/free"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _is_free(model: dict) -> bool:
|
|
23
|
+
model_id = model.get("id", "")
|
|
24
|
+
pricing = model.get("pricing", {})
|
|
25
|
+
return model_id.endswith(":free") or (
|
|
26
|
+
pricing.get("prompt") == "0" and pricing.get("completion") == "0"
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def fetch_free_models() -> list[dict]:
|
|
31
|
+
headers = {}
|
|
32
|
+
api_key = os.environ.get("OPENROUTER_API_KEY")
|
|
33
|
+
if api_key:
|
|
34
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
35
|
+
|
|
36
|
+
req = urllib.request.Request(MODELS_URL, headers=headers)
|
|
37
|
+
with urllib.request.urlopen(req, timeout=10) as resp:
|
|
38
|
+
payload = json.loads(resp.read().decode("utf-8"))
|
|
39
|
+
|
|
40
|
+
models = payload.get("data", [])
|
|
41
|
+
return [m for m in models if _is_free(m)]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def main() -> None:
|
|
45
|
+
print(f"RECOMMENDED (zero-maintenance default, auto-selects among current free models):\n {RECOMMENDED_MODEL}\n")
|
|
46
|
+
print("Set NARROW_MCP_SELECTOR_PROVIDER=openrouter and OPENROUTER_API_KEY, and you're done --")
|
|
47
|
+
print(f"the tool defaults to {RECOMMENDED_MODEL} automatically once an OPENROUTER_API_KEY is set.\n")
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
free_models = fetch_free_models()
|
|
51
|
+
except (urllib.error.URLError, TimeoutError, json.JSONDecodeError, OSError) as exc:
|
|
52
|
+
print(f"Could not fetch the live model list: {exc}", file=sys.stderr)
|
|
53
|
+
sys.exit(1)
|
|
54
|
+
|
|
55
|
+
if not free_models:
|
|
56
|
+
print("No individually-free models found right now (the meta-router above still works).")
|
|
57
|
+
return
|
|
58
|
+
|
|
59
|
+
print(f"Currently free, tool-calling-capable individual models ({len(free_models)}) --")
|
|
60
|
+
print("set NARROW_MCP_SELECTOR_MODEL=<id> to pick one specifically instead of the meta-router:\n")
|
|
61
|
+
id_width = max(len(m.get("id", "")) for m in free_models) + 2
|
|
62
|
+
print(f"{'id'.ljust(id_width)}context_length")
|
|
63
|
+
for m in sorted(free_models, key=lambda m: m.get("id", "")):
|
|
64
|
+
context_length = m.get("context_length", "?")
|
|
65
|
+
print(f"{m.get('id', '').ljust(id_width)}{context_length}")
|
|
66
|
+
|
|
67
|
+
print(
|
|
68
|
+
"\nNote: OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, "
|
|
69
|
+
"or 1000/day once the account has $10+ lifetime spend)."
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
if __name__ == "__main__":
|
|
74
|
+
main()
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Thin MCP adapter. No business logic lives here -- it registers the one
|
|
2
|
+
tool and calls straight into orchestrator.narrow_file, so the entire
|
|
3
|
+
pipeline is testable via plain Python calls with zero MCP machinery
|
|
4
|
+
involved.
|
|
5
|
+
|
|
6
|
+
Verified against the installed `mcp` package (v2.2.0, 2026-07-28 spec) in
|
|
7
|
+
this environment -- the v1 `FastMCP`/`@mcp.tool()` pattern referenced by
|
|
8
|
+
older tutorials no longer exists in this SDK major version; it was renamed
|
|
9
|
+
to `MCPServer` from `mcp.server.mcpserver`. If you're building against a
|
|
10
|
+
different SDK version, re-check this against `mcp.server.mcpserver`'s own
|
|
11
|
+
source before assuming this file's API calls still match.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from mcp.server.mcpserver import MCPServer
|
|
17
|
+
|
|
18
|
+
from .config import Config
|
|
19
|
+
from .models import ToolResult
|
|
20
|
+
from .orchestrator import narrow_file as _narrow_file
|
|
21
|
+
|
|
22
|
+
server = MCPServer(
|
|
23
|
+
"narrow-mcp",
|
|
24
|
+
instructions=(
|
|
25
|
+
"Narrows large, low-density non-source files (logs, build output, CSV, JSON, HTML) "
|
|
26
|
+
"down to the verbatim, line-numbered spans relevant to a stated intent -- verified "
|
|
27
|
+
"against the original file, never a summary. Refuses source code files; use "
|
|
28
|
+
"code-navigation tooling (LSP/tree-sitter/ast-grep) for those instead."
|
|
29
|
+
),
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@server.tool(
|
|
34
|
+
description=(
|
|
35
|
+
"Given a path to a large, low-density non-source file (logs, build output, CSV, JSON, "
|
|
36
|
+
"HTML) and a specific intent, returns only the verbatim line-numbered spans relevant to "
|
|
37
|
+
"that intent, verified against the original file -- never a summary. Refuses source code "
|
|
38
|
+
"files; use code-navigation tools for those."
|
|
39
|
+
)
|
|
40
|
+
)
|
|
41
|
+
def narrow_file(path: str, intent: str) -> ToolResult:
|
|
42
|
+
"""path: absolute or workspace-relative path to the target file.
|
|
43
|
+
|
|
44
|
+
intent: a specific question or goal, e.g. 'why did this build fail' or
|
|
45
|
+
'rows with null customer_id'. Required -- not inferred from context.
|
|
46
|
+
"""
|
|
47
|
+
config = Config.load()
|
|
48
|
+
return _narrow_file(path, intent, config)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def main() -> None:
|
|
52
|
+
server.run(transport="stdio")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
if __name__ == "__main__":
|
|
56
|
+
main()
|