llm-api-scope 0.4.0__tar.gz → 0.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/PKG-INFO +7 -2
  2. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/README.md +6 -1
  3. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/config.py +25 -1
  4. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/main.py +50 -5
  5. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/app.py +15 -10
  6. llm_api_scope-0.5.1/apiscope/openapi/fetch.py +58 -0
  7. llm_api_scope-0.5.1/apiscope/rfc/__init__.py +6 -0
  8. llm_api_scope-0.5.1/apiscope/rfc/app.py +403 -0
  9. llm_api_scope-0.5.1/apiscope/rfc/fetch.py +31 -0
  10. llm_api_scope-0.5.1/apiscope/rfc/parse_txt.py +54 -0
  11. llm_api_scope-0.5.1/apiscope/rfc/parse_xml.py +93 -0
  12. llm_api_scope-0.5.1/apiscope/rfc/schema.py +130 -0
  13. llm_api_scope-0.5.1/apiscope/rfc/search.py +60 -0
  14. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/schema.py +2 -0
  15. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/PKG-INFO +7 -2
  16. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/SOURCES.txt +7 -0
  17. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/pyproject.toml +1 -1
  18. llm_api_scope-0.4.0/apiscope/openapi/fetch.py +0 -49
  19. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/LICENSE +0 -0
  20. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/__init__.py +0 -0
  21. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/__init__.py +0 -0
  22. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/reader.py +0 -0
  23. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/schema.py +0 -0
  24. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/spec/__init__.py +0 -0
  25. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/spec/app.py +0 -0
  26. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/apiscope/openapi/spec/schema.py +0 -0
  27. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/dependency_links.txt +0 -0
  28. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/entry_points.txt +0 -0
  29. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/requires.txt +0 -0
  30. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/llm_api_scope.egg-info/top_level.txt +0 -0
  31. {llm_api_scope-0.4.0 → llm_api_scope-0.5.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.4.0
3
+ Version: 0.5.1
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -73,9 +73,14 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
73
73
 
74
74
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
75
75
 
76
+ ### rfc
77
+
78
+ read, search, and navigate RFC documents from the IETF.
79
+
80
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
81
+
76
82
  ## future
77
83
 
78
- - read RFC documents by number
79
84
  - read academic papers from arxiv
80
85
  - more formal document formats as the need arises
81
86
 
@@ -24,9 +24,14 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
24
24
 
25
25
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
26
26
 
27
+ ### rfc
28
+
29
+ read, search, and navigate RFC documents from the IETF.
30
+
31
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
32
+
27
33
  ## future
28
34
 
29
- - read RFC documents by number
30
35
  - read academic papers from arxiv
31
36
  - more formal document formats as the need arises
32
37
 
@@ -1,6 +1,7 @@
1
1
  # apiscope/config.py
2
2
 
3
3
  import json
4
+ import time
4
5
  from contextlib import contextmanager
5
6
  from os import environ
6
7
  from pathlib import Path
@@ -27,13 +28,36 @@ CACHE_ROOT = DEFAULT_ROOT / "cache"
27
28
  # ==============================================================================
28
29
 
29
30
 
30
- class OpenapiConfig(BaseModel):
31
+ class BaseConfig(BaseModel):
32
+ cache_ttl: int # seconds
33
+
34
+ def is_stale_since(self, timestamp: float) -> bool:
35
+ return time.time() - timestamp > self.cache_ttl
36
+
37
+ def is_stale_path(self, path: Path) -> bool:
38
+ if not path.exists():
39
+ return True
40
+ stat = path.stat()
41
+ latest = max(stat.st_mtime, stat.st_ctime)
42
+ return self.is_stale_since(latest)
43
+
44
+
45
+ class OpenapiConfig(BaseConfig):
31
46
  proxy: str | None = Field(default=None)
32
47
  alias: dict[str, str] = Field(default_factory=dict)
33
48
 
49
+ # override
50
+ cache_ttl: int = Field(default=60 * 60 * 24) # 1 day
51
+
52
+
53
+ class RfcConfig(BaseConfig):
54
+ # override
55
+ cache_ttl: int = Field(default=60 * 60 * 24 * 7) # 1 week
56
+
34
57
 
35
58
  class Config(BaseModel):
36
59
  openapi: OpenapiConfig = Field(default_factory=OpenapiConfig)
60
+ rfc: RfcConfig = Field(default_factory=RfcConfig)
37
61
 
38
62
  @classmethod
39
63
  def read(cls, path: Path) -> "Config":
@@ -12,10 +12,20 @@
12
12
  # subcommand callbacks extend ctx.obj.extras with group-specific keys.
13
13
  # schema.py exists solely to document the context shape — no runtime logic.
14
14
 
15
+ import json
16
+
15
17
  import typer
16
18
 
17
- from apiscope.config import APP_NAME, DEFAULT_ROOT, get_config
19
+ from apiscope.config import (
20
+ APP_NAME,
21
+ CACHE_ROOT,
22
+ DEFAULT_CONFIG_PATH,
23
+ DEFAULT_ROOT,
24
+ get_config,
25
+ get_project_config_path,
26
+ )
18
27
  from apiscope.openapi import openapi_app
28
+ from apiscope.rfc import check_rfc_deps, rfc_app
19
29
  from apiscope.schema import CommandContext
20
30
 
21
31
  # ==============================================================================
@@ -46,6 +56,7 @@ def callback(ctx: typer.Context) -> None:
46
56
  # ==============================================================================
47
57
 
48
58
  app.add_typer(openapi_app, name="openapi")
59
+ app.add_typer(rfc_app, name="rfc")
49
60
 
50
61
  # ==============================================================================
51
62
  # commands
@@ -53,11 +64,45 @@ app.add_typer(openapi_app, name="openapi")
53
64
 
54
65
 
55
66
  @app.command(help="check that apiscope is installed and working")
56
- def health(ctx: typer.Context) -> None:
57
- cache_dir = DEFAULT_ROOT / "cache"
67
+ def health(
68
+ ctx: typer.Context,
69
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
70
+ ) -> None:
71
+ # ensure directories exist
58
72
  DEFAULT_ROOT.mkdir(parents=True, exist_ok=True)
59
- cache_dir.mkdir(parents=True, exist_ok=True)
60
- typer.echo(f"{APP_NAME} is healthy")
73
+ CACHE_ROOT.mkdir(parents=True, exist_ok=True)
74
+
75
+ # gather issues from all modules
76
+ issues: list[str] = []
77
+ for label, check in [("rfc", check_rfc_deps)]:
78
+ err = check()
79
+ if err is not None:
80
+ issues.append(f"[{label}] {err}")
81
+
82
+ # config paths
83
+ global_cfg = str(DEFAULT_CONFIG_PATH) if DEFAULT_CONFIG_PATH.exists() else None
84
+ project_path = get_project_config_path()
85
+ project_cfg = str(project_path) if project_path and project_path.exists() else None
86
+
87
+ if json_output:
88
+ result: dict = {
89
+ "home": str(DEFAULT_ROOT),
90
+ "config": {"global": global_cfg, "project": project_cfg},
91
+ "cache": str(CACHE_ROOT),
92
+ "issues": issues,
93
+ }
94
+ typer.echo(json.dumps(result, ensure_ascii=False))
95
+ else:
96
+ typer.echo(f"[home] <{DEFAULT_ROOT}>")
97
+ typer.echo(f"[config] global <{global_cfg or 'none'}>")
98
+ typer.echo(f"[config] project <{project_cfg or 'none'}>")
99
+ typer.echo(f"[cache] <{CACHE_ROOT}>")
100
+ typer.echo("---")
101
+ if issues:
102
+ for msg in issues:
103
+ typer.echo(f"[!] {msg}", err=True)
104
+ raise typer.Exit(code=1)
105
+ typer.echo(f"[+] {APP_NAME} is healthy")
61
106
 
62
107
 
63
108
  # ==============================================================================
@@ -7,7 +7,7 @@ from typing import Any
7
7
  import typer
8
8
 
9
9
  from apiscope.config import CACHE_ROOT, Config
10
- from apiscope.openapi.fetch import fetch_openapi_spec
10
+ from apiscope.openapi.fetch import fetch_openapi_spec, openapi_cache_path
11
11
  from apiscope.openapi.reader import HttpMethod, OpenapiReader
12
12
  from apiscope.openapi.schema import OpenapiCommandContext
13
13
  from apiscope.openapi.spec import spec_app
@@ -26,16 +26,20 @@ def _resolve_source(alias_or_source: str, config: Config) -> str:
26
26
  return alias_or_source
27
27
 
28
28
 
29
- def _load_reader(source: str, cache_dir: Path, proxy: str | None = None) -> OpenapiReader:
30
- cached = fetch_openapi_spec(source, cache_dir, proxy)
29
+ def _load_reader(
30
+ source: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
31
+ ) -> OpenapiReader:
32
+ cached = fetch_openapi_spec(source, cache_dir, proxy, refresh=refresh)
31
33
  return OpenapiReader.load(cached)
32
34
 
33
35
 
34
- def _get_reader(source: str, ctx: typer.Context) -> OpenapiReader:
36
+ def _get_reader(source: str, ctx: typer.Context, force: bool = False) -> OpenapiReader:
35
37
  resolved = _resolve_source(source, ctx.obj.config)
36
38
  cache_dir = ctx.obj.openapi_command_context.cache_dir
37
39
  proxy = ctx.obj.config.openapi.proxy
38
- return _load_reader(resolved, cache_dir, proxy)
40
+ cache_path = openapi_cache_path(resolved, cache_dir)
41
+ stale = ctx.obj.config.openapi.is_stale_path(cache_path)
42
+ return _load_reader(resolved, cache_dir, proxy, refresh=force or stale)
39
43
 
40
44
 
41
45
  # ==============================================================================
@@ -61,9 +65,9 @@ def list_operations(
61
65
  source: str = typer.Argument(help="alias or path to the OpenAPI spec"),
62
66
  tag: str | None = typer.Option(default=None, help="filter by tag"),
63
67
  method: str | None = typer.Option(default=None, help="filter by HTTP method"),
68
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
64
69
  ) -> None:
65
- # load and resolve the spec
66
- reader = _get_reader(source, ctx)
70
+ reader = _get_reader(source, ctx, force=force)
67
71
 
68
72
  # collect all operations (path + method + identity fields)
69
73
  operations: list[dict[str, Any]] = []
@@ -106,9 +110,9 @@ def describe_operation(
106
110
  request: bool = typer.Option(
107
111
  default=False, show_default=False, help="show only request fields, omit responses"
108
112
  ),
113
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
109
114
  ) -> None:
110
- # load and resolve the spec
111
- reader = _get_reader(source, ctx)
115
+ reader = _get_reader(source, ctx, force=force)
112
116
 
113
117
  # merge path-item parameters with operation parameters
114
118
  path_item = reader.paths.get(path, {})
@@ -132,8 +136,9 @@ def describe_operation(
132
136
  def show_info(
133
137
  ctx: typer.Context,
134
138
  source: str = typer.Argument(help="alias or path to the OpenAPI spec"),
139
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
135
140
  ) -> None:
136
- reader = _get_reader(source, ctx)
141
+ reader = _get_reader(source, ctx, force=force)
137
142
 
138
143
  META_KEYS = ("openapi", "info", "servers", "tags", "security", "externalDocs")
139
144
  meta = {k: v for k, v in reader.raw.items() if k in META_KEYS}
@@ -0,0 +1,58 @@
1
+ # apiscope/openapi/fetch.py
2
+
3
+ import shutil
4
+ from hashlib import sha256
5
+ from pathlib import Path
6
+ from urllib.parse import unquote, urlparse
7
+
8
+ import httpx
9
+
10
+ OPENAPI_EXTENSIONS = {".json", ".yaml", ".yml"}
11
+
12
+
13
+ def _cache_key(source: str) -> str:
14
+ return sha256(source.encode()).hexdigest()
15
+
16
+
17
+ def openapi_cache_path(source: str, cache_dir: Path) -> Path:
18
+ if "://" in source:
19
+ path = unquote(urlparse(source).path).rstrip()
20
+ else:
21
+ path = source
22
+ suffix = Path(path).suffix.lower()
23
+ ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
24
+ return cache_dir / f"{_cache_key(source)}{ext}"
25
+
26
+
27
+ def _fetch_local(source: str, cache_dir: Path, refresh: bool = False) -> Path:
28
+ cache_path = openapi_cache_path(source, cache_dir)
29
+
30
+ if refresh or not cache_path.exists():
31
+ src = Path(source).expanduser().resolve()
32
+ shutil.copy2(src, cache_path)
33
+
34
+ return cache_path
35
+
36
+
37
+ def _fetch_remote(
38
+ url: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
39
+ ) -> Path:
40
+ cache_path = openapi_cache_path(url, cache_dir)
41
+
42
+ if refresh or not cache_path.exists():
43
+ client_kwargs: dict = {}
44
+ if proxy is not None:
45
+ client_kwargs["proxy"] = proxy
46
+ resp = httpx.get(url, follow_redirects=True, **client_kwargs)
47
+ resp.raise_for_status()
48
+ cache_path.write_bytes(resp.content)
49
+
50
+ return cache_path
51
+
52
+
53
+ def fetch_openapi_spec(
54
+ source: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
55
+ ) -> Path:
56
+ if "://" in source:
57
+ return _fetch_remote(source, cache_dir, proxy, refresh=refresh)
58
+ return _fetch_local(source, cache_dir, refresh=refresh)
@@ -0,0 +1,6 @@
1
+ # apiscope/rfc/__init__.py
2
+
3
+ from apiscope.rfc.app import app as rfc_app
4
+ from apiscope.rfc.app import check_deps as check_rfc_deps
5
+
6
+ __all__ = ["rfc_app", "check_rfc_deps"]
@@ -0,0 +1,403 @@
1
+ # apiscope/rfc/app.py
2
+
3
+ import json
4
+ import shutil
5
+
6
+ import typer
7
+
8
+ from apiscope.config import CACHE_ROOT
9
+ from apiscope.rfc.fetch import fetch_all_index_json, fetch_content_by_number
10
+ from apiscope.rfc.parse_xml import TocEntry
11
+ from apiscope.rfc.schema import ContentFormat, RfcCommandContext, RfcMetadata, RfcStatus
12
+ from apiscope.rfc.search import match_trigram, search_content
13
+
14
+ app = typer.Typer(help="browse IETF RFC documents")
15
+
16
+
17
+ # ==============================================================================
18
+ # public helpers
19
+ # ==============================================================================
20
+
21
+
22
+ def check_deps() -> str | None:
23
+ if shutil.which("rsync") is None:
24
+ return "rsync is required but not found in PATH"
25
+ return None
26
+
27
+
28
+ # ==============================================================================
29
+ # callback
30
+ # ==============================================================================
31
+
32
+
33
+ @app.callback()
34
+ def rfc_callback(ctx: typer.Context) -> None:
35
+ # check required external tools
36
+ err = check_deps()
37
+ if err is not None:
38
+ typer.echo(err, err=True)
39
+ raise typer.Exit(code=1)
40
+
41
+ # prepare cache directories
42
+ rfc_cache_dir = CACHE_ROOT / "rfc"
43
+ index_json_dir = rfc_cache_dir / "index"
44
+ content_xml_dir = rfc_cache_dir / "content" / "xml"
45
+ content_txt_dir = rfc_cache_dir / "content" / "txt"
46
+ index_json_dir.mkdir(parents=True, exist_ok=True)
47
+ content_xml_dir.mkdir(parents=True, exist_ok=True)
48
+ content_txt_dir.mkdir(parents=True, exist_ok=True)
49
+
50
+ # inject context for subcommands
51
+ ctx.obj.rfc_command_context = RfcCommandContext(
52
+ index_json_dir=index_json_dir,
53
+ content_xml_dir=content_xml_dir,
54
+ content_txt_dir=content_txt_dir,
55
+ )
56
+
57
+
58
+ # ==============================================================================
59
+ # helpers
60
+ # ==============================================================================
61
+
62
+
63
+ def _print_toc_entry(entry: TocEntry, indent: int = 0) -> None:
64
+ # root entry has no self line, skip directly to children
65
+ if entry.id:
66
+ prefix = " " * indent
67
+ typer.echo(f"{prefix}{entry.id}. {entry.title}")
68
+ child_indent = indent + (1 if entry.id else 0)
69
+ for child in entry.children:
70
+ _print_toc_entry(child, child_indent)
71
+
72
+
73
+ def _show_page_info(content: str, number: int, rfc_ctx: RfcCommandContext) -> None:
74
+ from apiscope.rfc.parse_txt import page_count
75
+
76
+ total = page_count(content)
77
+ path = rfc_ctx.get_content_txt_path(number)
78
+
79
+ typer.echo(f"[TXT] rfc {number} | pages: 1-{total}")
80
+ typer.echo("this document has no table of contents")
81
+ typer.echo("use --page <number> to read a specific page")
82
+ typer.echo(f"grep or ripgrep at <{path}>")
83
+
84
+
85
+ # ==============================================================================
86
+ # commands
87
+ # ==============================================================================
88
+
89
+
90
+ @app.command(name="sync", help="download RFC metadata index via rsync")
91
+ def sync_index(
92
+ ctx: typer.Context,
93
+ force: bool = typer.Option(False, "--force", help="force re-sync ignoring cache TTL"),
94
+ ) -> None:
95
+ rfc_ctx = ctx.obj.rfc_command_context
96
+ rfc_config = ctx.obj.config.rfc
97
+ index_dir = rfc_ctx.index_json_dir
98
+
99
+ if not force and not rfc_config.is_stale_path(index_dir):
100
+ typer.echo("index is up to date")
101
+ return
102
+
103
+ typer.echo("syncing via rsync...")
104
+ err = fetch_all_index_json(index_dir)
105
+ if err is not None:
106
+ typer.echo(f"sync failed\n{err}", err=True)
107
+ raise typer.Exit(code=1)
108
+ typer.echo("done")
109
+
110
+
111
+ @app.command(name="info", help="show RFC metadata")
112
+ def show_info(
113
+ ctx: typer.Context,
114
+ number: int = typer.Argument(help="RFC number"),
115
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
116
+ ) -> None:
117
+ rfc_ctx = ctx.obj.rfc_command_context
118
+ meta = rfc_ctx.get_index_json(number)
119
+ if meta is None:
120
+ typer.echo(f"rfc {number} not found", err=True)
121
+ raise typer.Exit(code=1)
122
+
123
+ # output raw JSON
124
+ if json_output:
125
+ typer.echo(meta.model_dump_json())
126
+ return
127
+
128
+ # human-readable entry
129
+ typer.echo(f"[RFC-{number}]")
130
+ for field_name, value in meta.to_info_data():
131
+ # null/empty → (null)
132
+ if value is None or (isinstance(value, str) and not value.strip()):
133
+ formatted = "(null)"
134
+ elif isinstance(value, list) and not value:
135
+ formatted = "(null)"
136
+ elif isinstance(value, list):
137
+ formatted = "[" + ", ".join(str(v) for v in value) + "]"
138
+ elif field_name == "abstract":
139
+ formatted = str(value).replace("\r\n", " ").replace("\n", " ")
140
+ else:
141
+ formatted = str(value)
142
+ typer.echo(f"{field_name}: {formatted}")
143
+
144
+
145
+ @app.command(name="read", help="show RFC content or table of contents")
146
+ def read_content(
147
+ ctx: typer.Context,
148
+ number: int = typer.Argument(help="RFC number"),
149
+ section: str | None = typer.Option(None, "--section", help="extract by section id (XML only)"),
150
+ page: int | None = typer.Option(None, "--page", help="extract by page number (TXT only)"),
151
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
152
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache TTL"),
153
+ ) -> None:
154
+ rfc_ctx = ctx.obj.rfc_command_context
155
+ rfc_config = ctx.obj.config.rfc
156
+ meta = rfc_ctx.get_index_json(number)
157
+ if meta is None:
158
+ typer.echo(f"rfc {number} not found", err=True)
159
+ raise typer.Exit(code=1)
160
+
161
+ # determine content format
162
+ if meta.is_xml_format:
163
+ fmt: ContentFormat = "xml"
164
+ content_path = rfc_ctx.get_content_xml_path(number)
165
+ elif meta.is_txt_format:
166
+ fmt = "txt"
167
+ content_path = rfc_ctx.get_content_txt_path(number)
168
+ else:
169
+ typer.echo(f"rfc {number} has no readable content format", err=True)
170
+ raise typer.Exit(code=1)
171
+
172
+ # ensure content is available locally
173
+ if force or rfc_config.is_stale_path(content_path):
174
+ typer.echo(f"fetching <{content_path.name}> via rsync...")
175
+ err = fetch_content_by_number(number, content_path.parent, fmt)
176
+ if err is not None:
177
+ typer.echo(f"failed to fetch content\n{err}", err=True)
178
+ raise typer.Exit(code=1)
179
+
180
+ content = content_path.read_text()
181
+
182
+ # dispatch by format
183
+ if section is not None and page is not None:
184
+ typer.echo("--section and --page are mutually exclusive", err=True)
185
+ raise typer.Exit(code=1)
186
+
187
+ if fmt == "xml":
188
+ if page is not None:
189
+ typer.echo("--page is not available for XML format", err=True)
190
+ raise typer.Exit(code=1)
191
+
192
+ from apiscope.rfc.parse_xml import parse_xml_section, parse_xml_toc
193
+
194
+ if section is not None:
195
+ entry = parse_xml_section(content, section)
196
+ if json_output:
197
+ typer.echo(entry.model_dump_json(indent=2))
198
+ elif entry.content is not None:
199
+ typer.echo(entry.content)
200
+ else:
201
+ _print_toc_entry(entry)
202
+ else:
203
+ tree = parse_xml_toc(content)
204
+ if json_output:
205
+ typer.echo(tree.model_dump_json(indent=2))
206
+ else:
207
+ _print_toc_entry(tree)
208
+ return
209
+
210
+ if fmt == "txt":
211
+ if section is not None:
212
+ typer.echo("--section is not available for TXT format", err=True)
213
+ typer.echo("use --page <number> to read a specific page", err=True)
214
+ raise typer.Exit(code=1)
215
+
216
+ from apiscope.rfc.parse_txt import extract_page
217
+
218
+ if page is not None:
219
+ try:
220
+ page_content = extract_page(content, page)
221
+ except ValueError:
222
+ typer.echo(f"page {page} not found in rfc {number}", err=True)
223
+ raise typer.Exit(code=1)
224
+ if json_output:
225
+ typer.echo(json.dumps({"page": page, "content": page_content}))
226
+ else:
227
+ typer.echo(page_content)
228
+ else:
229
+ _show_page_info(content, number, rfc_ctx)
230
+ return
231
+
232
+
233
+ @app.command(name="search", help="search RFC index or fulltext")
234
+ def search_rfc(
235
+ ctx: typer.Context,
236
+ number: int | None = typer.Argument(None, help="RFC number for fulltext search"),
237
+ term: str | None = typer.Argument(None, help="search term for fulltext search"),
238
+ status: list[str] | None = typer.Option(None, "--status", "-s", help="filter by RFC status"),
239
+ since: int | None = typer.Option(None, "--since", help="filter by start year"),
240
+ until: int | None = typer.Option(None, "--until", help="filter by end year"),
241
+ author: str | None = typer.Option(None, "--author", "-a", help="filter by author name"),
242
+ title: str | None = typer.Option(None, "--title", "-t", help="filter by title keyword"),
243
+ abstract: str | None = typer.Option(None, "--abstract", help="filter by abstract keyword"),
244
+ keywords_field: str | None = typer.Option(None, "--keywords", help="filter by keywords field"),
245
+ source: str | None = typer.Option(None, "--source", help="filter by source / working group"),
246
+ context: int = typer.Option(1, "--context", "-C", help="lines of context around match"),
247
+ no_snippet: bool = typer.Option(False, "--no-snippet", help="hide content snippet"),
248
+ limit: int = typer.Option(20, "--limit", help="max results shown"),
249
+ offset: int = typer.Option(0, "--offset", help="skip first N results"),
250
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache TTL"),
251
+ ) -> None:
252
+ rfc_ctx = ctx.obj.rfc_command_context
253
+ rfc_config = ctx.obj.config.rfc
254
+ # step 1: dispatch mode
255
+ index_filters = [status, since, until, author, title, abstract, keywords_field, source]
256
+ fulltext_mode = number is not None and term is not None
257
+ index_mode = number is None and term is None
258
+
259
+ if not fulltext_mode and not index_mode:
260
+ typer.echo(
261
+ "provide both NUMBER and TERM for fulltext search, or neither for index search",
262
+ err=True,
263
+ )
264
+ raise typer.Exit(code=1)
265
+
266
+ if fulltext_mode and any(index_filters):
267
+ typer.echo(
268
+ "filter flags only valid in index mode (search without a number)",
269
+ err=True,
270
+ )
271
+ raise typer.Exit(code=1)
272
+
273
+ if index_mode and (context != 1 or no_snippet):
274
+ typer.echo(
275
+ "--context and --no-snippet only valid in fulltext mode (search with a number)",
276
+ err=True,
277
+ )
278
+ raise typer.Exit(code=1)
279
+
280
+ # step 2: fulltext search
281
+ if fulltext_mode:
282
+ assert number is not None and term is not None
283
+ meta = rfc_ctx.get_index_json(number)
284
+ if meta is None:
285
+ typer.echo(f"rfc {number} not found", err=True)
286
+ raise typer.Exit(code=1)
287
+
288
+ content_path = rfc_ctx.get_content_txt_path(number)
289
+ if force or rfc_config.is_stale_path(content_path):
290
+ typer.echo(f"fetching <{content_path.name}> via rsync...")
291
+ err = fetch_content_by_number(number, content_path.parent, "txt")
292
+ if err is not None:
293
+ typer.echo(f"failed to fetch content\n{err}", err=True)
294
+ raise typer.Exit(code=1)
295
+
296
+ if not content_path.exists():
297
+ typer.echo(
298
+ f"rfc {number} has no TXT content, fulltext search unavailable",
299
+ err=True,
300
+ )
301
+ typer.echo("try reading with --section instead", err=True)
302
+ raise typer.Exit(code=1)
303
+
304
+ content = content_path.read_text()
305
+ page_matches = search_content(content, term, context)
306
+
307
+ if not page_matches:
308
+ typer.echo(f"no matches for '{term}' in rfc {number}")
309
+ return
310
+
311
+ total_ft = len(page_matches)
312
+ sliced_ft = page_matches[offset : offset + limit]
313
+
314
+ for page, snippet in sliced_ft:
315
+ if no_snippet:
316
+ typer.echo(f"[RFC-{number}] page {page}")
317
+ else:
318
+ typer.echo(f"[RFC-{number}] page {page}")
319
+ typer.echo(snippet)
320
+ typer.echo()
321
+
322
+ if total_ft > offset + limit:
323
+ shown = offset + limit
324
+ typer.echo(f"... ({shown} of {total_ft} matches shown, use --offset {shown} for more)")
325
+ return
326
+
327
+ # step 3: index search — scan all RFC metadata
328
+ rfc_matches = []
329
+
330
+ # validate status flags
331
+ wanted_status: set[RfcStatus] = set()
332
+ if status:
333
+ for raw in status:
334
+ s = RfcStatus.from_string(raw)
335
+ if s is None:
336
+ valid = ", ".join(f"{m.name.lower()}({m.value})" for m in RfcStatus)
337
+ typer.echo(f"unknown status '{raw}'. available: {valid}", err=True)
338
+ raise typer.Exit(code=1)
339
+ wanted_status.add(s)
340
+
341
+ for f in sorted(rfc_ctx.index_json_dir.glob("rfc*.json"), reverse=True):
342
+ meta = RfcMetadata.from_json_file(f)
343
+
344
+ # status filter
345
+ if wanted_status and meta.status not in wanted_status:
346
+ continue
347
+
348
+ # date range filter
349
+ if since is not None or until is not None:
350
+ pub_date = meta.pub_date or ""
351
+ year_str = pub_date.split()[-1] if pub_date else ""
352
+ try:
353
+ year = int(year_str)
354
+ except (ValueError, IndexError):
355
+ year = 0
356
+ if since is not None and year < since:
357
+ continue
358
+ if until is not None and year > until:
359
+ continue
360
+
361
+ # string filters via trigram
362
+ if author and not any(match_trigram(a, author) for a in meta.authors if a):
363
+ continue
364
+ if title and not match_trigram(meta.title or "", title):
365
+ continue
366
+ if abstract and not match_trigram(meta.abstract or "", abstract):
367
+ continue
368
+ if keywords_field and not any(match_trigram(k, keywords_field) for k in meta.keywords if k):
369
+ continue
370
+ if source and not match_trigram(meta.source or "", source):
371
+ continue
372
+
373
+ rfc_matches.append(meta)
374
+
375
+ # step 4: print index results
376
+ if not rfc_matches:
377
+ typer.echo("no matching RFCs found")
378
+ return
379
+
380
+ total = len(rfc_matches)
381
+ sliced = rfc_matches[offset : offset + limit]
382
+
383
+ for meta in sliced:
384
+ doc_id = (meta.doc_id or "").replace("RFC", "RFC-")
385
+ authors = meta.authors[:3]
386
+ authors_str = ", ".join(authors)
387
+ if len(meta.authors) > 3:
388
+ authors_str += ", et al."
389
+
390
+ abstract_text = (meta.abstract or "").replace("\r\n", " ").replace("\n", " ")
391
+ abstract_line = abstract_text.split(". ")[0] if abstract_text else ""
392
+ if abstract_line and not abstract_line.endswith("."):
393
+ abstract_line += "."
394
+
395
+ typer.echo(f"[{doc_id}] {meta.title or ''}")
396
+ typer.echo(f" {meta.status or ''} | {meta.pub_date or ''} | {authors_str}")
397
+ if abstract_line:
398
+ typer.echo(f" {abstract_line}")
399
+ typer.echo()
400
+
401
+ if total > offset + limit:
402
+ shown = offset + limit
403
+ typer.echo(f"... ({shown} of {total} results shown, use --offset {shown} for more)")
@@ -0,0 +1,31 @@
1
+ # apiscope/rfc/fetch.py
2
+
3
+ import subprocess
4
+ from pathlib import Path
5
+
6
+ from apiscope.rfc.schema import ContentFormat
7
+
8
+ # rsync module sources
9
+ RFC_RSYNC_HOST = "rsync.rfc-editor.org"
10
+ RFC_INDEX_MODULE = "rfcs-json-only"
11
+ RFC_CONTENT_MODULE = "rfcs"
12
+
13
+
14
+ def _do_rsync(src: str, dst: str) -> str | None:
15
+ result = subprocess.run(
16
+ ["rsync", "-az", "--delete", src, f"{dst}/"],
17
+ capture_output=True,
18
+ text=True,
19
+ )
20
+ if result.returncode != 0:
21
+ return result.stderr.strip()
22
+ return None
23
+
24
+
25
+ def fetch_all_index_json(dst_dir: Path) -> str | None:
26
+ return _do_rsync(f"{RFC_RSYNC_HOST}::{RFC_INDEX_MODULE}", str(dst_dir))
27
+
28
+
29
+ def fetch_content_by_number(number: int, dst_dir: Path, fmt: ContentFormat) -> str | None:
30
+ filename = f"rfc{number}.{fmt}"
31
+ return _do_rsync(f"{RFC_RSYNC_HOST}::{RFC_CONTENT_MODULE}/{filename}", str(dst_dir))
@@ -0,0 +1,54 @@
1
+ # apiscope/rfc/parse_txt.py
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ PAGE_RE = re.compile(r"\[Page (\d+)\]")
8
+
9
+
10
+ # ==============================================================================
11
+ # public api
12
+ # ==============================================================================
13
+
14
+
15
+ def page_count(content: str) -> int:
16
+ matches = PAGE_RE.findall(content)
17
+ return int(matches[-1]) if matches else 1
18
+
19
+
20
+ def split_pages(content: str) -> list[tuple[int, int, int]]:
21
+ # returns [(page_number, start_line, end_line), ...]
22
+ lines = content.split("\n")
23
+ boundaries: list[tuple[int, int]] = [] # [(page, line_index), ...]
24
+
25
+ for i, line in enumerate(lines):
26
+ m = PAGE_RE.search(line)
27
+ if m:
28
+ boundaries.append((int(m.group(1)), i))
29
+
30
+ if not boundaries:
31
+ return [(1, 0, len(lines))]
32
+
33
+ result: list[tuple[int, int, int]] = []
34
+ for idx, (page_num, line_num) in enumerate(boundaries):
35
+ start = 0 if idx == 0 else boundaries[idx - 1][1] + 1
36
+ end = line_num + 1
37
+ result.append((page_num, start, end))
38
+
39
+ # last page from last marker to EOF
40
+ last_end = boundaries[-1][1] + 1
41
+ if last_end < len(lines):
42
+ # assume next sequential page
43
+ next_page = boundaries[-1][0] + 1
44
+ result.append((next_page, last_end, len(lines)))
45
+
46
+ return result
47
+
48
+
49
+ def extract_page(content: str, page: int) -> str:
50
+ pages = split_pages(content)
51
+ for p, start, end in pages:
52
+ if p == page:
53
+ return "\n".join(content.split("\n")[start:end])
54
+ raise ValueError(f"page {page} not found")
@@ -0,0 +1,93 @@
1
+ # apiscope/rfc/parse_xml.py
2
+
3
+ from __future__ import annotations
4
+
5
+ import xml.etree.ElementTree as ET
6
+
7
+ from pydantic import BaseModel
8
+
9
+
10
+ class TocEntry(BaseModel):
11
+ # normalized section id: "3.5.1" "a.1" etc.
12
+ id: str
13
+ title: str
14
+ # non-None = leaf node with body text, children empty
15
+ # None = internal node, see children
16
+ content: str | None = None
17
+ children: list[TocEntry] = []
18
+
19
+
20
+ # ==============================================================================
21
+ # helpers
22
+ # ==============================================================================
23
+
24
+
25
+ def _pn_to_id(pn: str) -> str:
26
+ # pn = xml attribute: "section-3.5.1" "section-appendix.a.1"
27
+ if pn.startswith("section-appendix."):
28
+ return pn.replace("section-appendix.", "", 1)
29
+ if pn.startswith("section-"):
30
+ return pn.replace("section-", "", 1)
31
+ return pn
32
+
33
+
34
+ # ==============================================================================
35
+ # public api
36
+ # ==============================================================================
37
+
38
+
39
+ def parse_xml_toc(content: str) -> TocEntry:
40
+ tree = ET.fromstring(content)
41
+ children: list[TocEntry] = []
42
+
43
+ for parent_tag in ("middle", "back"):
44
+ parent = tree.find(parent_tag)
45
+ if parent is not None:
46
+ _collect_xml_toc(parent.findall("section"), children)
47
+
48
+ return TocEntry(id="", title="", children=children)
49
+
50
+
51
+ def _collect_xml_toc(sections: list[ET.Element], entries: list[TocEntry]) -> None:
52
+ for section in sections:
53
+ if section.get("toc") == "exclude":
54
+ continue
55
+
56
+ pn = section.get("pn", "")
57
+ section_id = _pn_to_id(pn)
58
+ name_elem = section.find("name")
59
+ title = "".join(name_elem.itertext()).strip() if name_elem is not None else ""
60
+
61
+ child_sections = section.findall("section")
62
+ children: list[TocEntry] = []
63
+ if child_sections:
64
+ _collect_xml_toc(child_sections, children)
65
+
66
+ entries.append(TocEntry(id=section_id, title=title, children=children))
67
+
68
+
69
+ def parse_xml_section(content: str, section_id: str) -> TocEntry:
70
+ tree = ET.fromstring(content)
71
+
72
+ pn = f"section-{section_id}"
73
+ section = tree.find(f".//*[@pn='{pn}']")
74
+ if section is None:
75
+ raise ValueError(f"section '{section_id}' not found")
76
+
77
+ name_elem = section.find("name")
78
+ title = "".join(name_elem.itertext()).strip() if name_elem is not None else ""
79
+
80
+ child_sections = section.findall("section")
81
+ if child_sections:
82
+ children: list[TocEntry] = []
83
+ _collect_xml_toc(child_sections, children)
84
+ return TocEntry(id=section_id, title=title, children=children)
85
+
86
+ # leaf: collect text from <t> elements
87
+ texts: list[str] = []
88
+ for t in section.iter("t"):
89
+ text = "".join(t.itertext()).strip()
90
+ if text:
91
+ texts.append(text)
92
+ body = "\n\n".join(texts)
93
+ return TocEntry(id=section_id, title=title, content=body)
@@ -0,0 +1,130 @@
1
+ # apiscope/rfc/schema.py
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import Enum
6
+ from pathlib import Path
7
+ from typing import Literal
8
+
9
+ from pydantic import BaseModel
10
+
11
+ INFO_FIELD_ORDER: list[str] = [
12
+ "title",
13
+ "authors",
14
+ "pub_status",
15
+ "status",
16
+ "pub_date",
17
+ "source",
18
+ "abstract",
19
+ "page_count",
20
+ "doi",
21
+ "draft",
22
+ "see_also",
23
+ "errata_url",
24
+ "obsoletes",
25
+ "obsoleted_by",
26
+ "updates",
27
+ "updated_by",
28
+ ]
29
+
30
+ ContentFormat = Literal["xml", "txt"]
31
+
32
+
33
+ class RfcStatus(str, Enum):
34
+ PS = "PROPOSED STANDARD"
35
+ DS = "DRAFT STANDARD"
36
+ STD = "INTERNET STANDARD"
37
+ BCP = "BEST CURRENT PRACTICE"
38
+ INFO = "INFORMATIONAL"
39
+ EXP = "EXPERIMENTAL"
40
+ HIST = "HISTORIC"
41
+ NOT_ISSUED = "NOT ISSUED"
42
+ UNKNOWN = "UNKNOWN"
43
+
44
+ def __str__(self) -> str:
45
+ return self.value
46
+
47
+ @classmethod
48
+ def from_string(cls, s: str) -> RfcStatus | None:
49
+ key = s.strip().upper()
50
+ for member in cls:
51
+ if member.name == key or member.value == key:
52
+ return member
53
+ return None
54
+
55
+
56
+ class RfcMetadata(BaseModel):
57
+ doc_id: str | None = None
58
+ title: str | None = None
59
+ authors: list[str] = []
60
+ pub_status: RfcStatus | None = None
61
+ status: RfcStatus | None = None
62
+ pub_date: str | None = None
63
+ abstract: str | None = None
64
+ keywords: list[str] = []
65
+ format: list[str] = []
66
+ source: str | None = None
67
+ page_count: str | None = None
68
+ doi: str | None = None
69
+ draft: str | None = None
70
+ see_also: list[str] = []
71
+ errata_url: str | None = None
72
+ obsoletes: list[str] = []
73
+ obsoleted_by: list[str] = []
74
+ updates: list[str] = []
75
+ updated_by: list[str] = []
76
+
77
+ @classmethod
78
+ def from_json_file(cls, path: Path) -> RfcMetadata:
79
+ return cls.model_validate_json(path.read_text())
80
+
81
+ def to_info_data(self) -> list[tuple[str, object]]:
82
+ dumped = self.model_dump()
83
+ return [(f, dumped[f]) for f in INFO_FIELD_ORDER]
84
+
85
+ def is_format(self, fmt: str) -> bool:
86
+ f = fmt.lower()
87
+ return any(f == mf.lower() for mf in self.format)
88
+
89
+ @property
90
+ def is_xml_format(self) -> bool:
91
+ return self.is_format("xml")
92
+
93
+ @property
94
+ def is_txt_format(self) -> bool:
95
+ return (
96
+ self.is_format("txt") or self.is_format("ascii") or not any(f for f in self.format if f)
97
+ )
98
+
99
+
100
+ class RfcCommandContext(BaseModel):
101
+ index_json_dir: Path
102
+ content_xml_dir: Path
103
+ content_txt_dir: Path
104
+
105
+ def get_index_json_path(self, number: int) -> Path:
106
+ return self.index_json_dir / f"rfc{number}.json"
107
+
108
+ def get_content_xml_path(self, number: int) -> Path:
109
+ return self.content_xml_dir / f"rfc{number}.xml"
110
+
111
+ def get_content_txt_path(self, number: int) -> Path:
112
+ return self.content_txt_dir / f"rfc{number}.txt"
113
+
114
+ def get_index_json(self, number: int) -> RfcMetadata | None:
115
+ path = self.get_index_json_path(number)
116
+ if not path.exists():
117
+ return None
118
+ return RfcMetadata.from_json_file(path)
119
+
120
+ def get_content_xml(self, number: int) -> str | None:
121
+ path = self.get_content_xml_path(number)
122
+ if not path.exists():
123
+ return None
124
+ return path.read_text()
125
+
126
+ def get_content_txt(self, number: int) -> str | None:
127
+ path = self.get_content_txt_path(number)
128
+ if not path.exists():
129
+ return None
130
+ return path.read_text()
@@ -0,0 +1,60 @@
1
+ # apiscope/rfc/search.py
2
+
3
+ from __future__ import annotations
4
+
5
+ from difflib import SequenceMatcher
6
+
7
+ from apiscope.rfc.parse_txt import split_pages
8
+
9
+
10
+ def match_trigram(text: str, keyword: str, threshold: float = 0.7) -> bool:
11
+ # check if keyword appears in text via trigram similarity
12
+ text_lower = text.lower()
13
+ kw_lower = keyword.lower()
14
+
15
+ if not kw_lower:
16
+ return False
17
+
18
+ if kw_lower in text_lower:
19
+ return True
20
+
21
+ # word-level fuzzy match for typos
22
+ for word in text_lower.split():
23
+ if len(word) < 3:
24
+ continue
25
+ if SequenceMatcher(None, kw_lower, word).ratio() >= threshold:
26
+ return True
27
+
28
+ return False
29
+
30
+
31
+ def search_content(content: str, keyword: str, context: int = 1) -> list[tuple[int, str]]:
32
+ pages = split_pages(content)
33
+ lines = content.split("\n")
34
+ kw_lower = keyword.lower()
35
+ results: list[tuple[int, str]] = []
36
+
37
+ for page, start, end in pages:
38
+ page_lines = lines[start:end]
39
+ match_idx = -1
40
+
41
+ for i, line in enumerate(page_lines):
42
+ stripped = line.strip("\x0c").strip()
43
+ if kw_lower in stripped.lower():
44
+ match_idx = i
45
+ break
46
+
47
+ if match_idx < 0:
48
+ continue
49
+
50
+ # extract context lines
51
+ ctx_start = max(0, match_idx - context)
52
+ ctx_end = min(len(page_lines), match_idx + context + 1)
53
+ snippet_lines = page_lines[ctx_start:ctx_end]
54
+
55
+ # strip form feed characters from snippet
56
+ snippet = "\n".join(snippet_lines).replace("\x0c", "").strip()
57
+ if snippet:
58
+ results.append((page, snippet))
59
+
60
+ return results
@@ -4,8 +4,10 @@ from pydantic import BaseModel
4
4
 
5
5
  from apiscope.config import Config
6
6
  from apiscope.openapi.schema import OpenapiCommandContext
7
+ from apiscope.rfc.schema import RfcCommandContext
7
8
 
8
9
 
9
10
  class CommandContext(BaseModel):
10
11
  config: Config
11
12
  openapi_command_context: OpenapiCommandContext | None = None
13
+ rfc_command_context: RfcCommandContext | None = None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.4.0
3
+ Version: 0.5.1
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -73,9 +73,14 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
73
73
 
74
74
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
75
75
 
76
+ ### rfc
77
+
78
+ read, search, and navigate RFC documents from the IETF.
79
+
80
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
81
+
76
82
  ## future
77
83
 
78
- - read RFC documents by number
79
84
  - read academic papers from arxiv
80
85
  - more formal document formats as the need arises
81
86
 
@@ -13,6 +13,13 @@ apiscope/openapi/schema.py
13
13
  apiscope/openapi/spec/__init__.py
14
14
  apiscope/openapi/spec/app.py
15
15
  apiscope/openapi/spec/schema.py
16
+ apiscope/rfc/__init__.py
17
+ apiscope/rfc/app.py
18
+ apiscope/rfc/fetch.py
19
+ apiscope/rfc/parse_txt.py
20
+ apiscope/rfc/parse_xml.py
21
+ apiscope/rfc/schema.py
22
+ apiscope/rfc/search.py
16
23
  llm_api_scope.egg-info/PKG-INFO
17
24
  llm_api_scope.egg-info/SOURCES.txt
18
25
  llm_api_scope.egg-info/dependency_links.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "llm-api-scope"
7
- version = "0.4.0"
7
+ version = "0.5.1"
8
8
  description = "read and cache structured documents from remote for LLM agents"
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
@@ -1,49 +0,0 @@
1
- # apiscope/openapi/fetch.py
2
-
3
- import shutil
4
- from hashlib import sha256
5
- from pathlib import Path
6
- from urllib.parse import unquote, urlparse
7
-
8
- import httpx
9
-
10
- OPENAPI_EXTENSIONS = {".json", ".yaml", ".yml"}
11
-
12
-
13
- def _cache_key(source: str) -> str:
14
- return sha256(source.encode()).hexdigest()
15
-
16
-
17
- def _fetch_local(source: str, cache_dir: Path) -> Path:
18
- suffix = Path(source).suffix.lower()
19
- ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
20
- cache_path = cache_dir / f"{_cache_key(source)}{ext}"
21
-
22
- if not cache_path.exists():
23
- src = Path(source).expanduser().resolve()
24
- shutil.copy2(src, cache_path)
25
-
26
- return cache_path
27
-
28
-
29
- def _fetch_remote(url: str, cache_dir: Path, proxy: str | None = None) -> Path:
30
- path = unquote(urlparse(url).path).rstrip()
31
- suffix = Path(path).suffix.lower()
32
- ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
33
- cache_path = cache_dir / f"{_cache_key(url)}{ext}"
34
-
35
- if not cache_path.exists():
36
- client_kwargs: dict = {}
37
- if proxy is not None:
38
- client_kwargs["proxy"] = proxy
39
- resp = httpx.get(url, follow_redirects=True, **client_kwargs)
40
- resp.raise_for_status()
41
- cache_path.write_bytes(resp.content)
42
-
43
- return cache_path
44
-
45
-
46
- def fetch_openapi_spec(source: str, cache_dir: Path, proxy: str | None = None) -> Path:
47
- if "://" in source:
48
- return _fetch_remote(source, cache_dir, proxy)
49
- return _fetch_local(source, cache_dir)
File without changes
File without changes