llm-api-scope 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {llm_api_scope-0.4.0/llm_api_scope.egg-info → llm_api_scope-0.5.0}/PKG-INFO +1 -1
  2. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/main.py +50 -5
  3. llm_api_scope-0.5.0/apiscope/rfc/__init__.py +6 -0
  4. llm_api_scope-0.5.0/apiscope/rfc/app.py +389 -0
  5. llm_api_scope-0.5.0/apiscope/rfc/fetch.py +31 -0
  6. llm_api_scope-0.5.0/apiscope/rfc/parse_txt.py +54 -0
  7. llm_api_scope-0.5.0/apiscope/rfc/parse_xml.py +93 -0
  8. llm_api_scope-0.5.0/apiscope/rfc/schema.py +130 -0
  9. llm_api_scope-0.5.0/apiscope/rfc/search.py +60 -0
  10. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/schema.py +2 -0
  11. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0/llm_api_scope.egg-info}/PKG-INFO +1 -1
  12. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/llm_api_scope.egg-info/SOURCES.txt +7 -0
  13. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/pyproject.toml +1 -1
  14. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/LICENSE +0 -0
  15. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/README.md +0 -0
  16. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/__init__.py +0 -0
  17. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/config.py +0 -0
  18. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/__init__.py +0 -0
  19. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/app.py +0 -0
  20. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/fetch.py +0 -0
  21. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/reader.py +0 -0
  22. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/schema.py +0 -0
  23. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/spec/__init__.py +0 -0
  24. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/spec/app.py +0 -0
  25. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/apiscope/openapi/spec/schema.py +0 -0
  26. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/llm_api_scope.egg-info/dependency_links.txt +0 -0
  27. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/llm_api_scope.egg-info/entry_points.txt +0 -0
  28. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/llm_api_scope.egg-info/requires.txt +0 -0
  29. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/llm_api_scope.egg-info/top_level.txt +0 -0
  30. {llm_api_scope-0.4.0 → llm_api_scope-0.5.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -12,10 +12,20 @@
12
12
  # subcommand callbacks extend ctx.obj.extras with group-specific keys.
13
13
  # schema.py exists solely to document the context shape — no runtime logic.
14
14
 
15
+ import json
16
+
15
17
  import typer
16
18
 
17
- from apiscope.config import APP_NAME, DEFAULT_ROOT, get_config
19
+ from apiscope.config import (
20
+ APP_NAME,
21
+ CACHE_ROOT,
22
+ DEFAULT_CONFIG_PATH,
23
+ DEFAULT_ROOT,
24
+ get_config,
25
+ get_project_config_path,
26
+ )
18
27
  from apiscope.openapi import openapi_app
28
+ from apiscope.rfc import check_rfc_deps, rfc_app
19
29
  from apiscope.schema import CommandContext
20
30
 
21
31
  # ==============================================================================
@@ -46,6 +56,7 @@ def callback(ctx: typer.Context) -> None:
46
56
  # ==============================================================================
47
57
 
48
58
  app.add_typer(openapi_app, name="openapi")
59
+ app.add_typer(rfc_app, name="rfc")
49
60
 
50
61
  # ==============================================================================
51
62
  # commands
@@ -53,11 +64,45 @@ app.add_typer(openapi_app, name="openapi")
53
64
 
54
65
 
55
66
  @app.command(help="check that apiscope is installed and working")
56
- def health(ctx: typer.Context) -> None:
57
- cache_dir = DEFAULT_ROOT / "cache"
67
+ def health(
68
+ ctx: typer.Context,
69
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
70
+ ) -> None:
71
+ # ensure directories exist
58
72
  DEFAULT_ROOT.mkdir(parents=True, exist_ok=True)
59
- cache_dir.mkdir(parents=True, exist_ok=True)
60
- typer.echo(f"{APP_NAME} is healthy")
73
+ CACHE_ROOT.mkdir(parents=True, exist_ok=True)
74
+
75
+ # gather issues from all modules
76
+ issues: list[str] = []
77
+ for label, check in [("rfc", check_rfc_deps)]:
78
+ err = check()
79
+ if err is not None:
80
+ issues.append(f"[{label}] {err}")
81
+
82
+ # config paths
83
+ global_cfg = str(DEFAULT_CONFIG_PATH) if DEFAULT_CONFIG_PATH.exists() else None
84
+ project_path = get_project_config_path()
85
+ project_cfg = str(project_path) if project_path and project_path.exists() else None
86
+
87
+ if json_output:
88
+ result: dict = {
89
+ "home": str(DEFAULT_ROOT),
90
+ "config": {"global": global_cfg, "project": project_cfg},
91
+ "cache": str(CACHE_ROOT),
92
+ "issues": issues,
93
+ }
94
+ typer.echo(json.dumps(result, ensure_ascii=False))
95
+ else:
96
+ typer.echo(f"[home] <{DEFAULT_ROOT}>")
97
+ typer.echo(f"[config] global <{global_cfg or 'none'}>")
98
+ typer.echo(f"[config] project <{project_cfg or 'none'}>")
99
+ typer.echo(f"[cache] <{CACHE_ROOT}>")
100
+ typer.echo("---")
101
+ if issues:
102
+ for msg in issues:
103
+ typer.echo(f"[!] {msg}", err=True)
104
+ raise typer.Exit(code=1)
105
+ typer.echo(f"[+] {APP_NAME} is healthy")
61
106
 
62
107
 
63
108
  # ==============================================================================
@@ -0,0 +1,6 @@
1
+ # apiscope/rfc/__init__.py
2
+
3
+ from apiscope.rfc.app import app as rfc_app
4
+ from apiscope.rfc.app import check_deps as check_rfc_deps
5
+
6
+ __all__ = ["rfc_app", "check_rfc_deps"]
@@ -0,0 +1,389 @@
1
+ # apiscope/rfc/app.py
2
+
3
+ import json
4
+ import shutil
5
+
6
+ import typer
7
+
8
+ from apiscope.config import CACHE_ROOT
9
+ from apiscope.rfc.fetch import fetch_all_index_json, fetch_content_by_number
10
+ from apiscope.rfc.parse_xml import TocEntry
11
+ from apiscope.rfc.schema import ContentFormat, RfcCommandContext, RfcMetadata, RfcStatus
12
+ from apiscope.rfc.search import match_trigram, search_content
13
+
14
+ app = typer.Typer(help="browse IETF RFC documents")
15
+
16
+
17
+ # ==============================================================================
18
+ # public helpers
19
+ # ==============================================================================
20
+
21
+
22
+ def check_deps() -> str | None:
23
+ if shutil.which("rsync") is None:
24
+ return "rsync is required but not found in PATH"
25
+ return None
26
+
27
+
28
+ # ==============================================================================
29
+ # callback
30
+ # ==============================================================================
31
+
32
+
33
+ @app.callback()
34
+ def rfc_callback(ctx: typer.Context) -> None:
35
+ # check required external tools
36
+ err = check_deps()
37
+ if err is not None:
38
+ typer.echo(err, err=True)
39
+ raise typer.Exit(code=1)
40
+
41
+ # prepare cache directories
42
+ rfc_cache_dir = CACHE_ROOT / "rfc"
43
+ index_json_dir = rfc_cache_dir / "index"
44
+ content_xml_dir = rfc_cache_dir / "content" / "xml"
45
+ content_txt_dir = rfc_cache_dir / "content" / "txt"
46
+ index_json_dir.mkdir(parents=True, exist_ok=True)
47
+ content_xml_dir.mkdir(parents=True, exist_ok=True)
48
+ content_txt_dir.mkdir(parents=True, exist_ok=True)
49
+
50
+ # inject context for subcommands
51
+ ctx.obj.rfc_command_context = RfcCommandContext(
52
+ index_json_dir=index_json_dir,
53
+ content_xml_dir=content_xml_dir,
54
+ content_txt_dir=content_txt_dir,
55
+ )
56
+
57
+
58
+ # ==============================================================================
59
+ # helpers
60
+ # ==============================================================================
61
+
62
+
63
+ def _print_toc_entry(entry: TocEntry, indent: int = 0) -> None:
64
+ # root entry has no self line, skip directly to children
65
+ if entry.id:
66
+ prefix = " " * indent
67
+ typer.echo(f"{prefix}{entry.id}. {entry.title}")
68
+ child_indent = indent + (1 if entry.id else 0)
69
+ for child in entry.children:
70
+ _print_toc_entry(child, child_indent)
71
+
72
+
73
+ def _show_page_info(content: str, number: int, rfc_ctx: RfcCommandContext) -> None:
74
+ from apiscope.rfc.parse_txt import page_count
75
+
76
+ total = page_count(content)
77
+ path = rfc_ctx.get_content_txt_path(number)
78
+
79
+ typer.echo(f"[TXT] rfc {number} | pages: 1-{total}")
80
+ typer.echo("this document has no table of contents")
81
+ typer.echo("use --page <number> to read a specific page")
82
+ typer.echo(f"grep or ripgrep at <{path}>")
83
+
84
+
85
+ # ==============================================================================
86
+ # commands
87
+ # ==============================================================================
88
+
89
+
90
+ @app.command(name="sync", help="download RFC metadata index via rsync")
91
+ def sync_index(ctx: typer.Context) -> None:
92
+ rfc_ctx = ctx.obj.rfc_command_context
93
+ typer.echo("syncing via rsync...")
94
+ err = fetch_all_index_json(rfc_ctx.index_json_dir)
95
+ if err is not None:
96
+ typer.echo(f"sync failed\n{err}", err=True)
97
+ raise typer.Exit(code=1)
98
+ typer.echo("done")
99
+
100
+
101
+ @app.command(name="info", help="show RFC metadata")
102
+ def show_info(
103
+ ctx: typer.Context,
104
+ number: int = typer.Argument(help="RFC number"),
105
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
106
+ ) -> None:
107
+ rfc_ctx = ctx.obj.rfc_command_context
108
+ meta = rfc_ctx.get_index_json(number)
109
+ if meta is None:
110
+ typer.echo(f"rfc {number} not found", err=True)
111
+ raise typer.Exit(code=1)
112
+
113
+ # output raw JSON
114
+ if json_output:
115
+ typer.echo(meta.model_dump_json())
116
+ return
117
+
118
+ # human-readable entry
119
+ typer.echo(f"[RFC-{number}]")
120
+ for field_name, value in meta.to_info_data():
121
+ # null/empty → (null)
122
+ if value is None or (isinstance(value, str) and not value.strip()):
123
+ formatted = "(null)"
124
+ elif isinstance(value, list) and not value:
125
+ formatted = "(null)"
126
+ elif isinstance(value, list):
127
+ formatted = "[" + ", ".join(str(v) for v in value) + "]"
128
+ elif field_name == "abstract":
129
+ formatted = str(value).replace("\r\n", " ").replace("\n", " ")
130
+ else:
131
+ formatted = str(value)
132
+ typer.echo(f"{field_name}: {formatted}")
133
+
134
+
135
+ @app.command(name="read", help="show RFC content or table of contents")
136
+ def read_content(
137
+ ctx: typer.Context,
138
+ number: int = typer.Argument(help="RFC number"),
139
+ section: str | None = typer.Option(None, "--section", help="extract by section id (XML only)"),
140
+ page: int | None = typer.Option(None, "--page", help="extract by page number (TXT only)"),
141
+ json_output: bool = typer.Option(False, "--json", help="output as JSON"),
142
+ ) -> None:
143
+ rfc_ctx = ctx.obj.rfc_command_context
144
+ meta = rfc_ctx.get_index_json(number)
145
+ if meta is None:
146
+ typer.echo(f"rfc {number} not found", err=True)
147
+ raise typer.Exit(code=1)
148
+
149
+ # determine content format
150
+ if meta.is_xml_format:
151
+ fmt: ContentFormat = "xml"
152
+ content_path = rfc_ctx.get_content_xml_path(number)
153
+ elif meta.is_txt_format:
154
+ fmt = "txt"
155
+ content_path = rfc_ctx.get_content_txt_path(number)
156
+ else:
157
+ typer.echo(f"rfc {number} has no readable content format", err=True)
158
+ raise typer.Exit(code=1)
159
+
160
+ # ensure content is available locally
161
+ if not content_path.exists():
162
+ typer.echo(f"fetching <{content_path.name}> via rsync...")
163
+ err = fetch_content_by_number(number, content_path.parent, fmt)
164
+ if err is not None:
165
+ typer.echo(f"failed to fetch content\n{err}", err=True)
166
+ raise typer.Exit(code=1)
167
+
168
+ content = content_path.read_text()
169
+
170
+ # dispatch by format
171
+ if section is not None and page is not None:
172
+ typer.echo("--section and --page are mutually exclusive", err=True)
173
+ raise typer.Exit(code=1)
174
+
175
+ if fmt == "xml":
176
+ if page is not None:
177
+ typer.echo("--page is not available for XML format", err=True)
178
+ raise typer.Exit(code=1)
179
+
180
+ from apiscope.rfc.parse_xml import parse_xml_section, parse_xml_toc
181
+
182
+ if section is not None:
183
+ entry = parse_xml_section(content, section)
184
+ if json_output:
185
+ typer.echo(entry.model_dump_json(indent=2))
186
+ elif entry.content is not None:
187
+ typer.echo(entry.content)
188
+ else:
189
+ _print_toc_entry(entry)
190
+ else:
191
+ tree = parse_xml_toc(content)
192
+ if json_output:
193
+ typer.echo(tree.model_dump_json(indent=2))
194
+ else:
195
+ _print_toc_entry(tree)
196
+ return
197
+
198
+ if fmt == "txt":
199
+ if section is not None:
200
+ typer.echo("--section is not available for TXT format", err=True)
201
+ typer.echo("use --page <number> to read a specific page", err=True)
202
+ raise typer.Exit(code=1)
203
+
204
+ from apiscope.rfc.parse_txt import extract_page
205
+
206
+ if page is not None:
207
+ try:
208
+ page_content = extract_page(content, page)
209
+ except ValueError:
210
+ typer.echo(f"page {page} not found in rfc {number}", err=True)
211
+ raise typer.Exit(code=1)
212
+ if json_output:
213
+ typer.echo(json.dumps({"page": page, "content": page_content}))
214
+ else:
215
+ typer.echo(page_content)
216
+ else:
217
+ _show_page_info(content, number, rfc_ctx)
218
+ return
219
+
220
+
221
+ @app.command(name="search", help="search RFC index or fulltext")
222
+ def search_rfc(
223
+ ctx: typer.Context,
224
+ number: int | None = typer.Argument(None, help="RFC number for fulltext search"),
225
+ term: str | None = typer.Argument(None, help="search term for fulltext search"),
226
+ status: list[str] | None = typer.Option(None, "--status", "-s", help="filter by RFC status"),
227
+ since: int | None = typer.Option(None, "--since", help="filter by start year"),
228
+ until: int | None = typer.Option(None, "--until", help="filter by end year"),
229
+ author: str | None = typer.Option(None, "--author", "-a", help="filter by author name"),
230
+ title: str | None = typer.Option(None, "--title", "-t", help="filter by title keyword"),
231
+ abstract: str | None = typer.Option(None, "--abstract", help="filter by abstract keyword"),
232
+ keywords_field: str | None = typer.Option(None, "--keywords", help="filter by keywords field"),
233
+ source: str | None = typer.Option(None, "--source", help="filter by source / working group"),
234
+ context: int = typer.Option(1, "--context", "-C", help="lines of context around match"),
235
+ no_snippet: bool = typer.Option(False, "--no-snippet", help="hide content snippet"),
236
+ limit: int = typer.Option(20, "--limit", help="max results shown"),
237
+ offset: int = typer.Option(0, "--offset", help="skip first N results"),
238
+ ) -> None:
239
+ rfc_ctx = ctx.obj.rfc_command_context
240
+ # step 1: dispatch mode
241
+ index_filters = [status, since, until, author, title, abstract, keywords_field, source]
242
+ fulltext_mode = number is not None and term is not None
243
+ index_mode = number is None and term is None
244
+
245
+ if not fulltext_mode and not index_mode:
246
+ typer.echo(
247
+ "provide both NUMBER and TERM for fulltext search, or neither for index search",
248
+ err=True,
249
+ )
250
+ raise typer.Exit(code=1)
251
+
252
+ if fulltext_mode and any(index_filters):
253
+ typer.echo(
254
+ "filter flags only valid in index mode (search without a number)",
255
+ err=True,
256
+ )
257
+ raise typer.Exit(code=1)
258
+
259
+ if index_mode and (context != 1 or no_snippet):
260
+ typer.echo(
261
+ "--context and --no-snippet only valid in fulltext mode (search with a number)",
262
+ err=True,
263
+ )
264
+ raise typer.Exit(code=1)
265
+
266
+ # step 2: fulltext search
267
+ if fulltext_mode:
268
+ assert number is not None and term is not None
269
+ meta = rfc_ctx.get_index_json(number)
270
+ if meta is None:
271
+ typer.echo(f"rfc {number} not found", err=True)
272
+ raise typer.Exit(code=1)
273
+
274
+ content_path = rfc_ctx.get_content_txt_path(number)
275
+ if not content_path.exists():
276
+ typer.echo(f"fetching <{content_path.name}> via rsync...")
277
+ err = fetch_content_by_number(number, content_path.parent, "txt")
278
+ if err is not None:
279
+ typer.echo(f"failed to fetch content\n{err}", err=True)
280
+ raise typer.Exit(code=1)
281
+
282
+ if not content_path.exists():
283
+ typer.echo(
284
+ f"rfc {number} has no TXT content, fulltext search unavailable",
285
+ err=True,
286
+ )
287
+ typer.echo("try reading with --section instead", err=True)
288
+ raise typer.Exit(code=1)
289
+
290
+ content = content_path.read_text()
291
+ page_matches = search_content(content, term, context)
292
+
293
+ if not page_matches:
294
+ typer.echo(f"no matches for '{term}' in rfc {number}")
295
+ return
296
+
297
+ total_ft = len(page_matches)
298
+ sliced_ft = page_matches[offset : offset + limit]
299
+
300
+ for page, snippet in sliced_ft:
301
+ if no_snippet:
302
+ typer.echo(f"[RFC-{number}] page {page}")
303
+ else:
304
+ typer.echo(f"[RFC-{number}] page {page}")
305
+ typer.echo(snippet)
306
+ typer.echo()
307
+
308
+ if total_ft > offset + limit:
309
+ shown = offset + limit
310
+ typer.echo(f"... ({shown} of {total_ft} matches shown, use --offset {shown} for more)")
311
+ return
312
+
313
+ # step 3: index search — scan all RFC metadata
314
+ rfc_matches = []
315
+
316
+ # validate status flags
317
+ wanted_status: set[RfcStatus] = set()
318
+ if status:
319
+ for raw in status:
320
+ s = RfcStatus.from_string(raw)
321
+ if s is None:
322
+ valid = ", ".join(f"{m.name.lower()}({m.value})" for m in RfcStatus)
323
+ typer.echo(f"unknown status '{raw}'. available: {valid}", err=True)
324
+ raise typer.Exit(code=1)
325
+ wanted_status.add(s)
326
+
327
+ for f in sorted(rfc_ctx.index_json_dir.glob("rfc*.json"), reverse=True):
328
+ meta = RfcMetadata.from_json_file(f)
329
+
330
+ # status filter
331
+ if wanted_status and meta.status not in wanted_status:
332
+ continue
333
+
334
+ # date range filter
335
+ if since is not None or until is not None:
336
+ pub_date = meta.pub_date or ""
337
+ year_str = pub_date.split()[-1] if pub_date else ""
338
+ try:
339
+ year = int(year_str)
340
+ except (ValueError, IndexError):
341
+ year = 0
342
+ if since is not None and year < since:
343
+ continue
344
+ if until is not None and year > until:
345
+ continue
346
+
347
+ # string filters via trigram
348
+ if author and not any(match_trigram(a, author) for a in meta.authors if a):
349
+ continue
350
+ if title and not match_trigram(meta.title or "", title):
351
+ continue
352
+ if abstract and not match_trigram(meta.abstract or "", abstract):
353
+ continue
354
+ if keywords_field and not any(match_trigram(k, keywords_field) for k in meta.keywords if k):
355
+ continue
356
+ if source and not match_trigram(meta.source or "", source):
357
+ continue
358
+
359
+ rfc_matches.append(meta)
360
+
361
+ # step 4: print index results
362
+ if not rfc_matches:
363
+ typer.echo("no matching RFCs found")
364
+ return
365
+
366
+ total = len(rfc_matches)
367
+ sliced = rfc_matches[offset : offset + limit]
368
+
369
+ for meta in sliced:
370
+ doc_id = (meta.doc_id or "").replace("RFC", "RFC-")
371
+ authors = meta.authors[:3]
372
+ authors_str = ", ".join(authors)
373
+ if len(meta.authors) > 3:
374
+ authors_str += ", et al."
375
+
376
+ abstract_text = (meta.abstract or "").replace("\r\n", " ").replace("\n", " ")
377
+ abstract_line = abstract_text.split(". ")[0] if abstract_text else ""
378
+ if abstract_line and not abstract_line.endswith("."):
379
+ abstract_line += "."
380
+
381
+ typer.echo(f"[{doc_id}] {meta.title or ''}")
382
+ typer.echo(f" {meta.status or ''} | {meta.pub_date or ''} | {authors_str}")
383
+ if abstract_line:
384
+ typer.echo(f" {abstract_line}")
385
+ typer.echo()
386
+
387
+ if total > offset + limit:
388
+ shown = offset + limit
389
+ typer.echo(f"... ({shown} of {total} results shown, use --offset {shown} for more)")
@@ -0,0 +1,31 @@
1
+ # apiscope/rfc/fetch.py
2
+
3
+ import subprocess
4
+ from pathlib import Path
5
+
6
+ from apiscope.rfc.schema import ContentFormat
7
+
8
+ # rsync module sources
9
+ RFC_RSYNC_HOST = "rsync.rfc-editor.org"
10
+ RFC_INDEX_MODULE = "rfcs-json-only"
11
+ RFC_CONTENT_MODULE = "rfcs"
12
+
13
+
14
+ def _do_rsync(src: str, dst: str) -> str | None:
15
+ result = subprocess.run(
16
+ ["rsync", "-az", "--delete", src, f"{dst}/"],
17
+ capture_output=True,
18
+ text=True,
19
+ )
20
+ if result.returncode != 0:
21
+ return result.stderr.strip()
22
+ return None
23
+
24
+
25
+ def fetch_all_index_json(dst_dir: Path) -> str | None:
26
+ return _do_rsync(f"{RFC_RSYNC_HOST}::{RFC_INDEX_MODULE}", str(dst_dir))
27
+
28
+
29
+ def fetch_content_by_number(number: int, dst_dir: Path, fmt: ContentFormat) -> str | None:
30
+ filename = f"rfc{number}.{fmt}"
31
+ return _do_rsync(f"{RFC_RSYNC_HOST}::{RFC_CONTENT_MODULE}/{filename}", str(dst_dir))
@@ -0,0 +1,54 @@
1
+ # apiscope/rfc/parse_txt.py
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ PAGE_RE = re.compile(r"\[Page (\d+)\]")
8
+
9
+
10
+ # ==============================================================================
11
+ # public api
12
+ # ==============================================================================
13
+
14
+
15
+ def page_count(content: str) -> int:
16
+ matches = PAGE_RE.findall(content)
17
+ return int(matches[-1]) if matches else 1
18
+
19
+
20
+ def split_pages(content: str) -> list[tuple[int, int, int]]:
21
+ # returns [(page_number, start_line, end_line), ...]
22
+ lines = content.split("\n")
23
+ boundaries: list[tuple[int, int]] = [] # [(page, line_index), ...]
24
+
25
+ for i, line in enumerate(lines):
26
+ m = PAGE_RE.search(line)
27
+ if m:
28
+ boundaries.append((int(m.group(1)), i))
29
+
30
+ if not boundaries:
31
+ return [(1, 0, len(lines))]
32
+
33
+ result: list[tuple[int, int, int]] = []
34
+ for idx, (page_num, line_num) in enumerate(boundaries):
35
+ start = 0 if idx == 0 else boundaries[idx - 1][1] + 1
36
+ end = line_num + 1
37
+ result.append((page_num, start, end))
38
+
39
+ # last page from last marker to EOF
40
+ last_end = boundaries[-1][1] + 1
41
+ if last_end < len(lines):
42
+ # assume next sequential page
43
+ next_page = boundaries[-1][0] + 1
44
+ result.append((next_page, last_end, len(lines)))
45
+
46
+ return result
47
+
48
+
49
+ def extract_page(content: str, page: int) -> str:
50
+ pages = split_pages(content)
51
+ for p, start, end in pages:
52
+ if p == page:
53
+ return "\n".join(content.split("\n")[start:end])
54
+ raise ValueError(f"page {page} not found")
@@ -0,0 +1,93 @@
1
+ # apiscope/rfc/parse_xml.py
2
+
3
+ from __future__ import annotations
4
+
5
+ import xml.etree.ElementTree as ET
6
+
7
+ from pydantic import BaseModel
8
+
9
+
10
+ class TocEntry(BaseModel):
11
+ # normalized section id: "3.5.1" "a.1" etc.
12
+ id: str
13
+ title: str
14
+ # non-None = leaf node with body text, children empty
15
+ # None = internal node, see children
16
+ content: str | None = None
17
+ children: list[TocEntry] = []
18
+
19
+
20
+ # ==============================================================================
21
+ # helpers
22
+ # ==============================================================================
23
+
24
+
25
+ def _pn_to_id(pn: str) -> str:
26
+ # pn = xml attribute: "section-3.5.1" "section-appendix.a.1"
27
+ if pn.startswith("section-appendix."):
28
+ return pn.replace("section-appendix.", "", 1)
29
+ if pn.startswith("section-"):
30
+ return pn.replace("section-", "", 1)
31
+ return pn
32
+
33
+
34
+ # ==============================================================================
35
+ # public api
36
+ # ==============================================================================
37
+
38
+
39
+ def parse_xml_toc(content: str) -> TocEntry:
40
+ tree = ET.fromstring(content)
41
+ children: list[TocEntry] = []
42
+
43
+ for parent_tag in ("middle", "back"):
44
+ parent = tree.find(parent_tag)
45
+ if parent is not None:
46
+ _collect_xml_toc(parent.findall("section"), children)
47
+
48
+ return TocEntry(id="", title="", children=children)
49
+
50
+
51
+ def _collect_xml_toc(sections: list[ET.Element], entries: list[TocEntry]) -> None:
52
+ for section in sections:
53
+ if section.get("toc") == "exclude":
54
+ continue
55
+
56
+ pn = section.get("pn", "")
57
+ section_id = _pn_to_id(pn)
58
+ name_elem = section.find("name")
59
+ title = "".join(name_elem.itertext()).strip() if name_elem is not None else ""
60
+
61
+ child_sections = section.findall("section")
62
+ children: list[TocEntry] = []
63
+ if child_sections:
64
+ _collect_xml_toc(child_sections, children)
65
+
66
+ entries.append(TocEntry(id=section_id, title=title, children=children))
67
+
68
+
69
+ def parse_xml_section(content: str, section_id: str) -> TocEntry:
70
+ tree = ET.fromstring(content)
71
+
72
+ pn = f"section-{section_id}"
73
+ section = tree.find(f".//*[@pn='{pn}']")
74
+ if section is None:
75
+ raise ValueError(f"section '{section_id}' not found")
76
+
77
+ name_elem = section.find("name")
78
+ title = "".join(name_elem.itertext()).strip() if name_elem is not None else ""
79
+
80
+ child_sections = section.findall("section")
81
+ if child_sections:
82
+ children: list[TocEntry] = []
83
+ _collect_xml_toc(child_sections, children)
84
+ return TocEntry(id=section_id, title=title, children=children)
85
+
86
+ # leaf: collect text from <t> elements
87
+ texts: list[str] = []
88
+ for t in section.iter("t"):
89
+ text = "".join(t.itertext()).strip()
90
+ if text:
91
+ texts.append(text)
92
+ body = "\n\n".join(texts)
93
+ return TocEntry(id=section_id, title=title, content=body)
@@ -0,0 +1,130 @@
1
+ # apiscope/rfc/schema.py
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import Enum
6
+ from pathlib import Path
7
+ from typing import Literal
8
+
9
+ from pydantic import BaseModel
10
+
11
+ INFO_FIELD_ORDER: list[str] = [
12
+ "title",
13
+ "authors",
14
+ "pub_status",
15
+ "status",
16
+ "pub_date",
17
+ "source",
18
+ "abstract",
19
+ "page_count",
20
+ "doi",
21
+ "draft",
22
+ "see_also",
23
+ "errata_url",
24
+ "obsoletes",
25
+ "obsoleted_by",
26
+ "updates",
27
+ "updated_by",
28
+ ]
29
+
30
+ ContentFormat = Literal["xml", "txt"]
31
+
32
+
33
+ class RfcStatus(str, Enum):
34
+ PS = "PROPOSED STANDARD"
35
+ DS = "DRAFT STANDARD"
36
+ STD = "INTERNET STANDARD"
37
+ BCP = "BEST CURRENT PRACTICE"
38
+ INFO = "INFORMATIONAL"
39
+ EXP = "EXPERIMENTAL"
40
+ HIST = "HISTORIC"
41
+ NOT_ISSUED = "NOT ISSUED"
42
+ UNKNOWN = "UNKNOWN"
43
+
44
+ def __str__(self) -> str:
45
+ return self.value
46
+
47
+ @classmethod
48
+ def from_string(cls, s: str) -> RfcStatus | None:
49
+ key = s.strip().upper()
50
+ for member in cls:
51
+ if member.name == key or member.value == key:
52
+ return member
53
+ return None
54
+
55
+
56
+ class RfcMetadata(BaseModel):
57
+ doc_id: str | None = None
58
+ title: str | None = None
59
+ authors: list[str] = []
60
+ pub_status: RfcStatus | None = None
61
+ status: RfcStatus | None = None
62
+ pub_date: str | None = None
63
+ abstract: str | None = None
64
+ keywords: list[str] = []
65
+ format: list[str] = []
66
+ source: str | None = None
67
+ page_count: str | None = None
68
+ doi: str | None = None
69
+ draft: str | None = None
70
+ see_also: list[str] = []
71
+ errata_url: str | None = None
72
+ obsoletes: list[str] = []
73
+ obsoleted_by: list[str] = []
74
+ updates: list[str] = []
75
+ updated_by: list[str] = []
76
+
77
+ @classmethod
78
+ def from_json_file(cls, path: Path) -> RfcMetadata:
79
+ return cls.model_validate_json(path.read_text())
80
+
81
+ def to_info_data(self) -> list[tuple[str, object]]:
82
+ dumped = self.model_dump()
83
+ return [(f, dumped[f]) for f in INFO_FIELD_ORDER]
84
+
85
+ def is_format(self, fmt: str) -> bool:
86
+ f = fmt.lower()
87
+ return any(f == mf.lower() for mf in self.format)
88
+
89
+ @property
90
+ def is_xml_format(self) -> bool:
91
+ return self.is_format("xml")
92
+
93
+ @property
94
+ def is_txt_format(self) -> bool:
95
+ return (
96
+ self.is_format("txt") or self.is_format("ascii") or not any(f for f in self.format if f)
97
+ )
98
+
99
+
100
+ class RfcCommandContext(BaseModel):
101
+ index_json_dir: Path
102
+ content_xml_dir: Path
103
+ content_txt_dir: Path
104
+
105
+ def get_index_json_path(self, number: int) -> Path:
106
+ return self.index_json_dir / f"rfc{number}.json"
107
+
108
+ def get_content_xml_path(self, number: int) -> Path:
109
+ return self.content_xml_dir / f"rfc{number}.xml"
110
+
111
+ def get_content_txt_path(self, number: int) -> Path:
112
+ return self.content_txt_dir / f"rfc{number}.txt"
113
+
114
+ def get_index_json(self, number: int) -> RfcMetadata | None:
115
+ path = self.get_index_json_path(number)
116
+ if not path.exists():
117
+ return None
118
+ return RfcMetadata.from_json_file(path)
119
+
120
+ def get_content_xml(self, number: int) -> str | None:
121
+ path = self.get_content_xml_path(number)
122
+ if not path.exists():
123
+ return None
124
+ return path.read_text()
125
+
126
+ def get_content_txt(self, number: int) -> str | None:
127
+ path = self.get_content_txt_path(number)
128
+ if not path.exists():
129
+ return None
130
+ return path.read_text()
@@ -0,0 +1,60 @@
1
+ # apiscope/rfc/search.py
2
+
3
+ from __future__ import annotations
4
+
5
+ from difflib import SequenceMatcher
6
+
7
+ from apiscope.rfc.parse_txt import split_pages
8
+
9
+
10
+ def match_trigram(text: str, keyword: str, threshold: float = 0.7) -> bool:
11
+ # check if keyword appears in text via trigram similarity
12
+ text_lower = text.lower()
13
+ kw_lower = keyword.lower()
14
+
15
+ if not kw_lower:
16
+ return False
17
+
18
+ if kw_lower in text_lower:
19
+ return True
20
+
21
+ # word-level fuzzy match for typos
22
+ for word in text_lower.split():
23
+ if len(word) < 3:
24
+ continue
25
+ if SequenceMatcher(None, kw_lower, word).ratio() >= threshold:
26
+ return True
27
+
28
+ return False
29
+
30
+
31
+ def search_content(content: str, keyword: str, context: int = 1) -> list[tuple[int, str]]:
32
+ pages = split_pages(content)
33
+ lines = content.split("\n")
34
+ kw_lower = keyword.lower()
35
+ results: list[tuple[int, str]] = []
36
+
37
+ for page, start, end in pages:
38
+ page_lines = lines[start:end]
39
+ match_idx = -1
40
+
41
+ for i, line in enumerate(page_lines):
42
+ stripped = line.strip("\x0c").strip()
43
+ if kw_lower in stripped.lower():
44
+ match_idx = i
45
+ break
46
+
47
+ if match_idx < 0:
48
+ continue
49
+
50
+ # extract context lines
51
+ ctx_start = max(0, match_idx - context)
52
+ ctx_end = min(len(page_lines), match_idx + context + 1)
53
+ snippet_lines = page_lines[ctx_start:ctx_end]
54
+
55
+ # strip form feed characters from snippet
56
+ snippet = "\n".join(snippet_lines).replace("\x0c", "").strip()
57
+ if snippet:
58
+ results.append((page, snippet))
59
+
60
+ return results
@@ -4,8 +4,10 @@ from pydantic import BaseModel
4
4
 
5
5
  from apiscope.config import Config
6
6
  from apiscope.openapi.schema import OpenapiCommandContext
7
+ from apiscope.rfc.schema import RfcCommandContext
7
8
 
8
9
 
9
10
  class CommandContext(BaseModel):
10
11
  config: Config
11
12
  openapi_command_context: OpenapiCommandContext | None = None
13
+ rfc_command_context: RfcCommandContext | None = None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -13,6 +13,13 @@ apiscope/openapi/schema.py
13
13
  apiscope/openapi/spec/__init__.py
14
14
  apiscope/openapi/spec/app.py
15
15
  apiscope/openapi/spec/schema.py
16
+ apiscope/rfc/__init__.py
17
+ apiscope/rfc/app.py
18
+ apiscope/rfc/fetch.py
19
+ apiscope/rfc/parse_txt.py
20
+ apiscope/rfc/parse_xml.py
21
+ apiscope/rfc/schema.py
22
+ apiscope/rfc/search.py
16
23
  llm_api_scope.egg-info/PKG-INFO
17
24
  llm_api_scope.egg-info/SOURCES.txt
18
25
  llm_api_scope.egg-info/dependency_links.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "llm-api-scope"
7
- version = "0.4.0"
7
+ version = "0.5.0"
8
8
  description = "read and cache structured documents from remote for LLM agents"
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
File without changes
File without changes
File without changes