narrow-mcp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Manik Prakash
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,126 @@
1
+ Metadata-Version: 2.4
2
+ Name: narrow-mcp
3
+ Version: 0.1.0
4
+ Summary: MCP server that narrows large low-density files (logs, CSV, JSON, HTML) to the verbatim spans relevant to an intent, verified against source.
5
+ Author: Manik Prakash
6
+ License-Expression: MIT
7
+ Project-URL: Repository, https://github.com/manik-prakash/narrow-mcp
8
+ Project-URL: Issues, https://github.com/manik-prakash/narrow-mcp/issues
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Topic :: Software Development :: Build Tools
17
+ Requires-Python: >=3.11
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ Requires-Dist: mcp[cli]>=2.0.0
21
+ Requires-Dist: anthropic>=0.40.0
22
+ Requires-Dist: openai>=1.0.0
23
+ Provides-Extra: dev
24
+ Requires-Dist: pytest>=8.0.0; extra == "dev"
25
+ Requires-Dist: pytest-asyncio>=0.24.0; extra == "dev"
26
+ Requires-Dist: pyyaml>=6.0; extra == "dev"
27
+ Dynamic: license-file
28
+
29
+ # narrow-mcp
30
+
31
+ <!-- mcp-name: io.github.manik-prakash/narrow-mcp -->
32
+
33
+ An MCP server that narrows large, low-density non-source files (logs, test
34
+ output, CSV, JSON, HTML) down to the verbatim spans relevant to a stated
35
+ intent -- verified against the original file, never a summary.
36
+
37
+ ## Why
38
+
39
+ Coding agents burn context reading large low-density files. A 40k-token log
40
+ might hold a few hundred tokens of signal. This tool:
41
+
42
+ 1. Runs deterministic narrowing first (grep-style search, structural
43
+ parsing per file type, sampling) -- free, fast, zero LLM cost.
44
+ 2. If that alone resolves the query with confidence (e.g. an exact CSV
45
+ "null column" query), returns it directly. No LLM call at all.
46
+ 3. Otherwise passes only the narrowed candidates (never the raw file) to a
47
+ cheap, fast selector model that returns line ranges and a one-line
48
+ reason -- never prose.
49
+ 4. Re-reads the chosen line ranges from the **original file on disk** and
50
+ returns that verbatim text. The selector's own words are never trusted
51
+ or returned -- only its line-number coordinates, which get verified.
52
+ 5. If the selector fails, times out, or returns something invalid, falls
53
+ back to the deterministic candidate set rather than failing outright.
54
+
55
+ Non-goals: source code retrieval (use LSP/tree-sitter/ast-grep), prose
56
+ summarization, local/self-hosted models.
57
+
58
+ ## Status
59
+
60
+ v1, single file per call. Four file types: log/build-output, CSV, JSON
61
+ (single document or JSONL), HTML.
62
+
63
+ ## Development
64
+
65
+ ```
66
+ pip install -e ".[dev]"
67
+ pytest
68
+ python eval/run_eval.py # mocked selector, free
69
+ python eval/run_eval.py --live # real selector call, needs an API key (see Configuration)
70
+ ```
71
+
72
+ ## Configuration
73
+
74
+ **You only need to set one API key.** The provider is auto-detected from
75
+ whichever key is present -- no separate provider/model config required:
76
+
77
+ | If you set... | Provider used | Default model |
78
+ |----------------------|----------------|-------------------------------------------------|
79
+ | `ANTHROPIC_API_KEY` | `anthropic` | `claude-haiku-4-5` |
80
+ | `OPENAI_API_KEY` | `openai` | `gpt-5-nano` |
81
+ | `OPENROUTER_API_KEY` | `openrouter` | `openrouter/free` (see below) |
82
+ | *(none)* | `anthropic` | `claude-haiku-4-5` (calls just always fall back to the deterministic path) |
83
+
84
+ If more than one key is set, priority is Anthropic > OpenAI > OpenRouter.
85
+ Override anything explicitly with the env vars below.
86
+
87
+ ### Using OpenRouter's free models
88
+
89
+ OpenRouter still requires its own API key even for $0-cost models -- set
90
+ `OPENROUTER_API_KEY` and you're done, no other config needed. It defaults to
91
+ **`openrouter/free`**, a meta-router that auto-picks among whichever
92
+ tool-calling-capable models are currently free, so it never goes stale the
93
+ way hardcoding one specific `:free` model name would.
94
+
95
+ To see the current free-model roster live (it rotates) and pick a specific
96
+ one instead of the meta-router:
97
+
98
+ ```
99
+ narrow-mcp-list-free-models
100
+ ```
101
+
102
+ Then set `NARROW_MCP_SELECTOR_MODEL=<id>` to whichever one you want.
103
+
104
+ OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, or 1000/day
105
+ once the account has $10+ lifetime spend) -- fine for interactive use, worth
106
+ knowing about for batch runs.
107
+
108
+ ### All environment variables
109
+
110
+ - `NARROW_MCP_SELECTOR_PROVIDER` -- `anthropic` | `openai` | `openrouter`.
111
+ Overrides auto-detection.
112
+ - `NARROW_MCP_SELECTOR_MODEL` -- overrides the provider's default model.
113
+ - `NARROW_MCP_SELECTOR_API_KEY_ENV` -- overrides which env var holds the key.
114
+ - `NARROW_MCP_SELECTOR_TIMEOUT_S` (default `3.0`)
115
+ - `NARROW_MCP_MAX_CANDIDATE_CHARS`, `NARROW_MCP_MAX_CANDIDATES`,
116
+ `NARROW_MCP_CONTEXT_LINES`, `NARROW_MCP_RIPGREP_PATH`,
117
+ `NARROW_MCP_MAX_JSON_BYTES`
118
+
119
+ ## Registering with Claude Code
120
+
121
+ ```
122
+ claude mcp add narrow-mcp -- uvx narrow-mcp
123
+ ```
124
+
125
+ Verify current `claude mcp add` flag syntax against `claude mcp add --help`
126
+ before relying on the above -- CLI flags change across releases.
@@ -0,0 +1,98 @@
1
+ # narrow-mcp
2
+
3
+ <!-- mcp-name: io.github.manik-prakash/narrow-mcp -->
4
+
5
+ An MCP server that narrows large, low-density non-source files (logs, test
6
+ output, CSV, JSON, HTML) down to the verbatim spans relevant to a stated
7
+ intent -- verified against the original file, never a summary.
8
+
9
+ ## Why
10
+
11
+ Coding agents burn context reading large low-density files. A 40k-token log
12
+ might hold a few hundred tokens of signal. This tool:
13
+
14
+ 1. Runs deterministic narrowing first (grep-style search, structural
15
+ parsing per file type, sampling) -- free, fast, zero LLM cost.
16
+ 2. If that alone resolves the query with confidence (e.g. an exact CSV
17
+ "null column" query), returns it directly. No LLM call at all.
18
+ 3. Otherwise passes only the narrowed candidates (never the raw file) to a
19
+ cheap, fast selector model that returns line ranges and a one-line
20
+ reason -- never prose.
21
+ 4. Re-reads the chosen line ranges from the **original file on disk** and
22
+ returns that verbatim text. The selector's own words are never trusted
23
+ or returned -- only its line-number coordinates, which get verified.
24
+ 5. If the selector fails, times out, or returns something invalid, falls
25
+ back to the deterministic candidate set rather than failing outright.
26
+
27
+ Non-goals: source code retrieval (use LSP/tree-sitter/ast-grep), prose
28
+ summarization, local/self-hosted models.
29
+
30
+ ## Status
31
+
32
+ v1, single file per call. Four file types: log/build-output, CSV, JSON
33
+ (single document or JSONL), HTML.
34
+
35
+ ## Development
36
+
37
+ ```
38
+ pip install -e ".[dev]"
39
+ pytest
40
+ python eval/run_eval.py # mocked selector, free
41
+ python eval/run_eval.py --live # real selector call, needs an API key (see Configuration)
42
+ ```
43
+
44
+ ## Configuration
45
+
46
+ **You only need to set one API key.** The provider is auto-detected from
47
+ whichever key is present -- no separate provider/model config required:
48
+
49
+ | If you set... | Provider used | Default model |
50
+ |----------------------|----------------|-------------------------------------------------|
51
+ | `ANTHROPIC_API_KEY` | `anthropic` | `claude-haiku-4-5` |
52
+ | `OPENAI_API_KEY` | `openai` | `gpt-5-nano` |
53
+ | `OPENROUTER_API_KEY` | `openrouter` | `openrouter/free` (see below) |
54
+ | *(none)* | `anthropic` | `claude-haiku-4-5` (calls just always fall back to the deterministic path) |
55
+
56
+ If more than one key is set, priority is Anthropic > OpenAI > OpenRouter.
57
+ Override anything explicitly with the env vars below.
58
+
59
+ ### Using OpenRouter's free models
60
+
61
+ OpenRouter still requires its own API key even for $0-cost models -- set
62
+ `OPENROUTER_API_KEY` and you're done, no other config needed. It defaults to
63
+ **`openrouter/free`**, a meta-router that auto-picks among whichever
64
+ tool-calling-capable models are currently free, so it never goes stale the
65
+ way hardcoding one specific `:free` model name would.
66
+
67
+ To see the current free-model roster live (it rotates) and pick a specific
68
+ one instead of the meta-router:
69
+
70
+ ```
71
+ narrow-mcp-list-free-models
72
+ ```
73
+
74
+ Then set `NARROW_MCP_SELECTOR_MODEL=<id>` to whichever one you want.
75
+
76
+ OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, or 1000/day
77
+ once the account has $10+ lifetime spend) -- fine for interactive use, worth
78
+ knowing about for batch runs.
79
+
80
+ ### All environment variables
81
+
82
+ - `NARROW_MCP_SELECTOR_PROVIDER` -- `anthropic` | `openai` | `openrouter`.
83
+ Overrides auto-detection.
84
+ - `NARROW_MCP_SELECTOR_MODEL` -- overrides the provider's default model.
85
+ - `NARROW_MCP_SELECTOR_API_KEY_ENV` -- overrides which env var holds the key.
86
+ - `NARROW_MCP_SELECTOR_TIMEOUT_S` (default `3.0`)
87
+ - `NARROW_MCP_MAX_CANDIDATE_CHARS`, `NARROW_MCP_MAX_CANDIDATES`,
88
+ `NARROW_MCP_CONTEXT_LINES`, `NARROW_MCP_RIPGREP_PATH`,
89
+ `NARROW_MCP_MAX_JSON_BYTES`
90
+
91
+ ## Registering with Claude Code
92
+
93
+ ```
94
+ claude mcp add narrow-mcp -- uvx narrow-mcp
95
+ ```
96
+
97
+ Verify current `claude mcp add` flag syntax against `claude mcp add --help`
98
+ before relying on the above -- CLI flags change across releases.
@@ -0,0 +1,50 @@
1
+ [project]
2
+ name = "narrow-mcp"
3
+ version = "0.1.0"
4
+ description = "MCP server that narrows large low-density files (logs, CSV, JSON, HTML) to the verbatim spans relevant to an intent, verified against source."
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ license = "MIT"
8
+ license-files = ["LICENSE"]
9
+ authors = [{ name = "Manik Prakash" }]
10
+ classifiers = [
11
+ "Development Status :: 4 - Beta",
12
+ "Intended Audience :: Developers",
13
+ "Programming Language :: Python :: 3",
14
+ "Programming Language :: Python :: 3.11",
15
+ "Programming Language :: Python :: 3.12",
16
+ "Programming Language :: Python :: 3.13",
17
+ "Operating System :: OS Independent",
18
+ "Topic :: Software Development :: Build Tools",
19
+ ]
20
+ dependencies = [
21
+ "mcp[cli]>=2.0.0",
22
+ "anthropic>=0.40.0",
23
+ "openai>=1.0.0",
24
+ ]
25
+
26
+ [project.urls]
27
+ Repository = "https://github.com/manik-prakash/narrow-mcp"
28
+ Issues = "https://github.com/manik-prakash/narrow-mcp/issues"
29
+
30
+ [project.scripts]
31
+ narrow-mcp = "narrow_mcp.mcp_handler:main"
32
+ narrow-mcp-list-free-models = "narrow_mcp.list_free_models:main"
33
+
34
+ [project.optional-dependencies]
35
+ dev = ["pytest>=8.0.0", "pytest-asyncio>=0.24.0", "pyyaml>=6.0"]
36
+
37
+ [build-system]
38
+ requires = ["setuptools>=77.0.0"]
39
+ build-backend = "setuptools.build_meta"
40
+
41
+ [tool.setuptools.packages.find]
42
+ where = ["src"]
43
+
44
+ [tool.setuptools.package-data]
45
+ narrow_mcp = ["narrowing/keyword_hints.toml"]
46
+
47
+ [tool.pytest.ini_options]
48
+ pythonpath = ["src"]
49
+ testpaths = ["tests"]
50
+ asyncio_mode = "auto"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1 @@
1
+ """narrow_mcp: narrows large low-density files to the spans relevant to an intent."""
@@ -0,0 +1,39 @@
1
+ """Candidate-budget enforcement. Standalone so policy can be tuned without touching narrowing logic."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from .config import Budget
6
+ from .models import Candidate
7
+
8
+
9
+ def enforce(candidates: list[Candidate], budget: Budget) -> tuple[list[Candidate], int]:
10
+ """Truncate/prioritize candidates to fit the budget.
11
+
12
+ Returns (kept_candidates, truncated_count). Always keeps at least one
13
+ candidate if any were given. Prioritizes by score (highest first), then
14
+ by document order for ties.
15
+ """
16
+ if not candidates:
17
+ return [], 0
18
+
19
+ ordered = sorted(
20
+ enumerate(candidates), key=lambda pair: (-pair[1].score, pair[0])
21
+ )
22
+
23
+ kept: list[Candidate] = []
24
+ total_chars = 0
25
+ for _, candidate in ordered:
26
+ preview_len = len(candidate.preview_text)
27
+ if kept and (
28
+ len(kept) >= budget.max_candidates
29
+ or total_chars + preview_len > budget.max_candidate_chars
30
+ ):
31
+ continue
32
+ kept.append(candidate)
33
+ total_chars += preview_len
34
+
35
+ # Restore original document order for readability in the prompt.
36
+ kept_ids = {id(c) for c in kept}
37
+ kept_in_order = [c for c in candidates if id(c) in kept_ids]
38
+ truncated_count = len(candidates) - len(kept_in_order)
39
+ return kept_in_order, truncated_count
@@ -0,0 +1,78 @@
1
+ """All tunables live here. Loaded from environment variables with sane defaults."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from dataclasses import dataclass
7
+
8
+
9
+ @dataclass
10
+ class Budget:
11
+ max_candidate_chars: int = 8000
12
+ max_candidates: int = 30
13
+ max_context_lines_per_hit: int = 3
14
+
15
+
16
+ @dataclass
17
+ class SelectorConfig:
18
+ provider: str = "anthropic"
19
+ model: str = "claude-haiku-4-5"
20
+ api_key_env_var: str = "ANTHROPIC_API_KEY"
21
+ timeout_s: float = 3.0
22
+ max_retries: int = 0
23
+
24
+
25
+ _PROVIDER_DEFAULTS: dict[str, dict[str, str]] = {
26
+ "anthropic": {"model": "claude-haiku-4-5", "api_key_env_var": "ANTHROPIC_API_KEY"},
27
+ "openai": {"model": "gpt-5-nano", "api_key_env_var": "OPENAI_API_KEY"},
28
+ # openrouter/free is a stable meta-router that auto-picks among currently
29
+ # free, tool-calling-capable models -- never goes stale the way a specific
30
+ # ":free" model id would as OpenRouter's free roster rotates.
31
+ "openrouter": {"model": "openrouter/free", "api_key_env_var": "OPENROUTER_API_KEY"},
32
+ }
33
+
34
+
35
+ def _autodetect_provider() -> str:
36
+ """Only used when NARROW_MCP_SELECTOR_PROVIDER is not explicitly set.
37
+ Priority: a directly-paid-for native provider before OpenRouter (the
38
+ $0-cost fallback, not the preferred option if the user already has a
39
+ native key). No key at all -> today's unchanged default.
40
+ """
41
+ if os.environ.get("ANTHROPIC_API_KEY"):
42
+ return "anthropic"
43
+ if os.environ.get("OPENAI_API_KEY"):
44
+ return "openai"
45
+ if os.environ.get("OPENROUTER_API_KEY"):
46
+ return "openrouter"
47
+ return "anthropic"
48
+
49
+
50
+ @dataclass
51
+ class Config:
52
+ budget: Budget
53
+ selector: SelectorConfig
54
+ ripgrep_path: str | None = None
55
+ max_json_bytes: int = 20_000_000
56
+
57
+ @classmethod
58
+ def load(cls) -> "Config":
59
+ budget = Budget(
60
+ max_candidate_chars=int(os.environ.get("NARROW_MCP_MAX_CANDIDATE_CHARS", 8000)),
61
+ max_candidates=int(os.environ.get("NARROW_MCP_MAX_CANDIDATES", 30)),
62
+ max_context_lines_per_hit=int(os.environ.get("NARROW_MCP_CONTEXT_LINES", 3)),
63
+ )
64
+ provider = os.environ.get("NARROW_MCP_SELECTOR_PROVIDER") or _autodetect_provider()
65
+ defaults = _PROVIDER_DEFAULTS.get(provider, _PROVIDER_DEFAULTS["anthropic"])
66
+ selector = SelectorConfig(
67
+ provider=provider,
68
+ model=os.environ.get("NARROW_MCP_SELECTOR_MODEL", defaults["model"]),
69
+ api_key_env_var=os.environ.get("NARROW_MCP_SELECTOR_API_KEY_ENV", defaults["api_key_env_var"]),
70
+ timeout_s=float(os.environ.get("NARROW_MCP_SELECTOR_TIMEOUT_S", 3.0)),
71
+ max_retries=int(os.environ.get("NARROW_MCP_SELECTOR_MAX_RETRIES", 0)),
72
+ )
73
+ return cls(
74
+ budget=budget,
75
+ selector=selector,
76
+ ripgrep_path=os.environ.get("NARROW_MCP_RIPGREP_PATH"),
77
+ max_json_bytes=int(os.environ.get("NARROW_MCP_MAX_JSON_BYTES", 20_000_000)),
78
+ )
@@ -0,0 +1,113 @@
1
+ """Decide which narrowing strategy applies to a file, and refuse source/binary files.
2
+
3
+ This is the single place the "no source-code retrieval" non-goal is enforced.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import csv
9
+ import io
10
+ import json
11
+ from pathlib import Path
12
+
13
+ from .models import FileTypeDecision
14
+
15
+ SOURCE_EXTENSIONS = {
16
+ ".py", ".pyi", ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs",
17
+ ".go", ".rs", ".java", ".kt", ".kts", ".c", ".h", ".cpp", ".hpp", ".cc",
18
+ ".cs", ".rb", ".php", ".swift", ".scala", ".m", ".mm", ".sh", ".ps1",
19
+ ".sql", ".lua", ".r", ".pl", ".ex", ".exs", ".clj", ".hs", ".dart",
20
+ ".vue", ".svelte",
21
+ }
22
+
23
+ CSV_EXTENSIONS = {".csv", ".tsv"}
24
+ JSON_EXTENSIONS = {".json", ".jsonl", ".ndjson"}
25
+ HTML_EXTENSIONS = {".html", ".htm", ".xhtml"}
26
+ LOG_EXTENSIONS = {".log", ".txt", ".out", ".err"}
27
+
28
+ SNIFF_BYTES = 8192
29
+
30
+
31
+ def _is_binary(sample: bytes) -> bool:
32
+ return b"\x00" in sample
33
+
34
+
35
+ def _sniff_json(sample: bytes) -> bool:
36
+ text = sample.decode("utf-8", errors="ignore").strip()
37
+ if not text:
38
+ return False
39
+ first = text[0]
40
+ if first not in "{[":
41
+ # Could still be JSONL where first line starts with { or [.
42
+ first_line = text.splitlines()[0].strip() if text.splitlines() else ""
43
+ if not first_line or first_line[0] not in "{[":
44
+ return False
45
+ try:
46
+ # Try parsing just the first line as a JSON value (JSONL check) or
47
+ # the whole sample as a prefix of a larger JSON document.
48
+ first_line = text.splitlines()[0]
49
+ json.loads(first_line)
50
+ return True
51
+ except (json.JSONDecodeError, ValueError):
52
+ # Might be a large single JSON document where the first line alone
53
+ # doesn't parse (e.g. pretty-printed). Fall back to bracket check.
54
+ return first in "{["
55
+
56
+
57
+ def _sniff_csv(sample: bytes) -> bool:
58
+ text = sample.decode("utf-8", errors="ignore")
59
+ if not text.strip():
60
+ return False
61
+ try:
62
+ csv.Sniffer().sniff(text, delimiters=",;\t|")
63
+ return True
64
+ except csv.Error:
65
+ return False
66
+
67
+
68
+ def _sniff_html(sample: bytes) -> bool:
69
+ text = sample.decode("utf-8", errors="ignore").lstrip()
70
+ lowered = text.lower()
71
+ return lowered.startswith("<!doctype html") or lowered.startswith("<html") or "<body" in lowered[:2000]
72
+
73
+
74
+ def detect(path: Path) -> FileTypeDecision:
75
+ if not path.exists() or not path.is_file():
76
+ return FileTypeDecision(kind="unknown", confidence=0.0, reason="path does not exist or is not a file")
77
+
78
+ ext = path.suffix.lower()
79
+
80
+ if ext in SOURCE_EXTENSIONS:
81
+ return FileTypeDecision(
82
+ kind="source_refused",
83
+ confidence=1.0,
84
+ reason="source file — use LSP/tree-sitter/ast-grep tooling instead",
85
+ )
86
+
87
+ try:
88
+ with open(path, "rb") as f:
89
+ sample = f.read(SNIFF_BYTES)
90
+ except OSError as exc:
91
+ return FileTypeDecision(kind="unknown", confidence=0.0, reason=f"could not read file: {exc}")
92
+
93
+ if _is_binary(sample):
94
+ return FileTypeDecision(kind="binary_refused", confidence=1.0, reason="binary file not supported")
95
+
96
+ if ext in JSON_EXTENSIONS:
97
+ return FileTypeDecision(kind="json", confidence=1.0, reason=f"extension {ext}")
98
+ if ext in CSV_EXTENSIONS:
99
+ return FileTypeDecision(kind="csv", confidence=1.0, reason=f"extension {ext}")
100
+ if ext in HTML_EXTENSIONS:
101
+ return FileTypeDecision(kind="html", confidence=1.0, reason=f"extension {ext}")
102
+ if ext in LOG_EXTENSIONS:
103
+ return FileTypeDecision(kind="log", confidence=0.9, reason=f"extension {ext}")
104
+
105
+ # No recognized extension — sniff content.
106
+ if _sniff_json(sample):
107
+ return FileTypeDecision(kind="json", confidence=0.7, reason="content sniff: JSON-like")
108
+ if _sniff_html(sample):
109
+ return FileTypeDecision(kind="html", confidence=0.7, reason="content sniff: HTML-like")
110
+ if _sniff_csv(sample):
111
+ return FileTypeDecision(kind="csv", confidence=0.6, reason="content sniff: CSV-like")
112
+
113
+ return FileTypeDecision(kind="log", confidence=0.4, reason="no structured format detected, treating as text/log")
@@ -0,0 +1,74 @@
1
+ """Standalone setup-time helper -- NOT an MCP tool, not wired into the
2
+ server. Prints OpenRouter's currently-free, tool-calling-capable models,
3
+ fetched live (the free roster rotates, so this is never hardcoded).
4
+
5
+ Usage:
6
+ narrow-mcp-list-free-models
7
+ python -m narrow_mcp.list_free_models
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import os
14
+ import sys
15
+ import urllib.error
16
+ import urllib.request
17
+
18
+ MODELS_URL = "https://openrouter.ai/api/v1/models?supported_parameters=tools"
19
+ RECOMMENDED_MODEL = "openrouter/free"
20
+
21
+
22
+ def _is_free(model: dict) -> bool:
23
+ model_id = model.get("id", "")
24
+ pricing = model.get("pricing", {})
25
+ return model_id.endswith(":free") or (
26
+ pricing.get("prompt") == "0" and pricing.get("completion") == "0"
27
+ )
28
+
29
+
30
+ def fetch_free_models() -> list[dict]:
31
+ headers = {}
32
+ api_key = os.environ.get("OPENROUTER_API_KEY")
33
+ if api_key:
34
+ headers["Authorization"] = f"Bearer {api_key}"
35
+
36
+ req = urllib.request.Request(MODELS_URL, headers=headers)
37
+ with urllib.request.urlopen(req, timeout=10) as resp:
38
+ payload = json.loads(resp.read().decode("utf-8"))
39
+
40
+ models = payload.get("data", [])
41
+ return [m for m in models if _is_free(m)]
42
+
43
+
44
+ def main() -> None:
45
+ print(f"RECOMMENDED (zero-maintenance default, auto-selects among current free models):\n {RECOMMENDED_MODEL}\n")
46
+ print("Set NARROW_MCP_SELECTOR_PROVIDER=openrouter and OPENROUTER_API_KEY, and you're done --")
47
+ print(f"the tool defaults to {RECOMMENDED_MODEL} automatically once an OPENROUTER_API_KEY is set.\n")
48
+
49
+ try:
50
+ free_models = fetch_free_models()
51
+ except (urllib.error.URLError, TimeoutError, json.JSONDecodeError, OSError) as exc:
52
+ print(f"Could not fetch the live model list: {exc}", file=sys.stderr)
53
+ sys.exit(1)
54
+
55
+ if not free_models:
56
+ print("No individually-free models found right now (the meta-router above still works).")
57
+ return
58
+
59
+ print(f"Currently free, tool-calling-capable individual models ({len(free_models)}) --")
60
+ print("set NARROW_MCP_SELECTOR_MODEL=<id> to pick one specifically instead of the meta-router:\n")
61
+ id_width = max(len(m.get("id", "")) for m in free_models) + 2
62
+ print(f"{'id'.ljust(id_width)}context_length")
63
+ for m in sorted(free_models, key=lambda m: m.get("id", "")):
64
+ context_length = m.get("context_length", "?")
65
+ print(f"{m.get('id', '').ljust(id_width)}{context_length}")
66
+
67
+ print(
68
+ "\nNote: OpenRouter's free tier is rate-limited (20 req/min; 50 req/day, "
69
+ "or 1000/day once the account has $10+ lifetime spend)."
70
+ )
71
+
72
+
73
+ if __name__ == "__main__":
74
+ main()
@@ -0,0 +1,56 @@
1
+ """Thin MCP adapter. No business logic lives here -- it registers the one
2
+ tool and calls straight into orchestrator.narrow_file, so the entire
3
+ pipeline is testable via plain Python calls with zero MCP machinery
4
+ involved.
5
+
6
+ Verified against the installed `mcp` package (v2.2.0, 2026-07-28 spec) in
7
+ this environment -- the v1 `FastMCP`/`@mcp.tool()` pattern referenced by
8
+ older tutorials no longer exists in this SDK major version; it was renamed
9
+ to `MCPServer` from `mcp.server.mcpserver`. If you're building against a
10
+ different SDK version, re-check this against `mcp.server.mcpserver`'s own
11
+ source before assuming this file's API calls still match.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from mcp.server.mcpserver import MCPServer
17
+
18
+ from .config import Config
19
+ from .models import ToolResult
20
+ from .orchestrator import narrow_file as _narrow_file
21
+
22
+ server = MCPServer(
23
+ "narrow-mcp",
24
+ instructions=(
25
+ "Narrows large, low-density non-source files (logs, build output, CSV, JSON, HTML) "
26
+ "down to the verbatim, line-numbered spans relevant to a stated intent -- verified "
27
+ "against the original file, never a summary. Refuses source code files; use "
28
+ "code-navigation tooling (LSP/tree-sitter/ast-grep) for those instead."
29
+ ),
30
+ )
31
+
32
+
33
+ @server.tool(
34
+ description=(
35
+ "Given a path to a large, low-density non-source file (logs, build output, CSV, JSON, "
36
+ "HTML) and a specific intent, returns only the verbatim line-numbered spans relevant to "
37
+ "that intent, verified against the original file -- never a summary. Refuses source code "
38
+ "files; use code-navigation tools for those."
39
+ )
40
+ )
41
+ def narrow_file(path: str, intent: str) -> ToolResult:
42
+ """path: absolute or workspace-relative path to the target file.
43
+
44
+ intent: a specific question or goal, e.g. 'why did this build fail' or
45
+ 'rows with null customer_id'. Required -- not inferred from context.
46
+ """
47
+ config = Config.load()
48
+ return _narrow_file(path, intent, config)
49
+
50
+
51
+ def main() -> None:
52
+ server.run(transport="stdio")
53
+
54
+
55
+ if __name__ == "__main__":
56
+ main()