llm-api-scope 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {llm_api_scope-0.5.0/llm_api_scope.egg-info → llm_api_scope-0.6.0}/PKG-INFO +13 -2
  2. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/README.md +12 -1
  3. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/config.py +44 -1
  4. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/main.py +3 -1
  5. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/app.py +15 -10
  6. llm_api_scope-0.6.0/apiscope/openapi/fetch.py +58 -0
  7. llm_api_scope-0.6.0/apiscope/repo/__init__.py +6 -0
  8. llm_api_scope-0.6.0/apiscope/repo/app.py +152 -0
  9. llm_api_scope-0.6.0/apiscope/repo/fetch.py +132 -0
  10. llm_api_scope-0.6.0/apiscope/repo/schema.py +53 -0
  11. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/app.py +18 -4
  12. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/schema.py +2 -0
  13. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0/llm_api_scope.egg-info}/PKG-INFO +13 -2
  14. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/llm_api_scope.egg-info/SOURCES.txt +4 -0
  15. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/pyproject.toml +1 -1
  16. llm_api_scope-0.5.0/apiscope/openapi/fetch.py +0 -49
  17. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/LICENSE +0 -0
  18. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/__init__.py +0 -0
  19. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/__init__.py +0 -0
  20. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/reader.py +0 -0
  21. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/schema.py +0 -0
  22. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/spec/__init__.py +0 -0
  23. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/spec/app.py +0 -0
  24. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/openapi/spec/schema.py +0 -0
  25. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/__init__.py +0 -0
  26. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/fetch.py +0 -0
  27. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/parse_txt.py +0 -0
  28. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/parse_xml.py +0 -0
  29. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/schema.py +0 -0
  30. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/apiscope/rfc/search.py +0 -0
  31. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/llm_api_scope.egg-info/dependency_links.txt +0 -0
  32. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/llm_api_scope.egg-info/entry_points.txt +0 -0
  33. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/llm_api_scope.egg-info/requires.txt +0 -0
  34. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/llm_api_scope.egg-info/top_level.txt +0 -0
  35. {llm_api_scope-0.5.0 → llm_api_scope-0.6.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -73,9 +73,20 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
73
73
 
74
74
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
75
75
 
76
+ ### rfc
77
+
78
+ read, search, and navigate RFC documents from the IETF.
79
+
80
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
81
+
82
+ ### repo
83
+
84
+ sync documentation directories from any git repository to a local cache.
85
+
86
+ register repositories by URL with a target subdirectory and a ref — a branch, tag, or commit. the sync command clones with shallow depth, blobless filter, and sparse checkout so only the needed tree and files come over the wire. extracted docs land in the cache under a stable hash path, and a configurable TTL avoids redundant re-fetches. provider-agnostic, zero auth.
87
+
76
88
  ## future
77
89
 
78
- - read RFC documents by number
79
90
  - read academic papers from arxiv
80
91
  - more formal document formats as the need arises
81
92
 
@@ -24,9 +24,20 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
24
24
 
25
25
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
26
26
 
27
+ ### rfc
28
+
29
+ read, search, and navigate RFC documents from the IETF.
30
+
31
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
32
+
33
+ ### repo
34
+
35
+ sync documentation directories from any git repository to a local cache.
36
+
37
+ register repositories by URL with a target subdirectory and a ref — a branch, tag, or commit. the sync command clones with shallow depth, blobless filter, and sparse checkout so only the needed tree and files come over the wire. extracted docs land in the cache under a stable hash path, and a configurable TTL avoids redundant re-fetches. provider-agnostic, zero auth.
38
+
27
39
  ## future
28
40
 
29
- - read RFC documents by number
30
41
  - read academic papers from arxiv
31
42
  - more formal document formats as the need arises
32
43
 
@@ -1,6 +1,7 @@
1
1
  # apiscope/config.py
2
2
 
3
3
  import json
4
+ import time
4
5
  from contextlib import contextmanager
5
6
  from os import environ
6
7
  from pathlib import Path
@@ -22,18 +23,59 @@ DEFAULT_CONFIG_SCHEMA_PATH = DEFAULT_ROOT / "config.schema.json"
22
23
 
23
24
  CACHE_ROOT = DEFAULT_ROOT / "cache"
24
25
 
26
+ TMP_ROOT = DEFAULT_ROOT / "tmp"
27
+
25
28
  # ==============================================================================
26
29
  # config model
27
30
  # ==============================================================================
28
31
 
29
32
 
30
- class OpenapiConfig(BaseModel):
33
+ class BaseConfig(BaseModel):
34
+ cache_ttl: int # seconds
35
+
36
+ def is_stale_since(self, timestamp: float) -> bool:
37
+ return time.time() - timestamp > self.cache_ttl
38
+
39
+ def is_stale_path(self, path: Path) -> bool:
40
+ if not path.exists():
41
+ return True
42
+ stat = path.stat()
43
+ latest = max(stat.st_mtime, stat.st_ctime)
44
+ return self.is_stale_since(latest)
45
+
46
+
47
+ class OpenapiConfig(BaseConfig):
31
48
  proxy: str | None = Field(default=None)
32
49
  alias: dict[str, str] = Field(default_factory=dict)
33
50
 
51
+ # override
52
+ cache_ttl: int = Field(default=60 * 60 * 24) # 1 day
53
+
54
+
55
+ class RfcConfig(BaseConfig):
56
+ # override
57
+ cache_ttl: int = Field(default=60 * 60 * 24 * 30) # 1 month
58
+
59
+
60
+ class RepoEntryConfig(BaseModel):
61
+ dir: str
62
+
63
+ # format - branch:<branch>, tag:<tag>, commit:<short-hash>
64
+ # e.g. branch:main, tag:v1.0, commit:abc1234
65
+ target: str = Field(default="branch:main")
66
+
67
+
68
+ class RepoConfig(BaseConfig):
69
+ entries: dict[str, RepoEntryConfig] = Field(default_factory=dict)
70
+
71
+ # override
72
+ cache_ttl: int = Field(default=60 * 60 * 24 * 7) # 1 week
73
+
34
74
 
35
75
  class Config(BaseModel):
36
76
  openapi: OpenapiConfig = Field(default_factory=OpenapiConfig)
77
+ rfc: RfcConfig = Field(default_factory=RfcConfig)
78
+ repo: RepoConfig = Field(default_factory=RepoConfig)
37
79
 
38
80
  @classmethod
39
81
  def read(cls, path: Path) -> "Config":
@@ -61,6 +103,7 @@ class Config(BaseModel):
61
103
  merged.openapi.alias |= other.openapi.alias
62
104
  if other.openapi.proxy is not None:
63
105
  merged.openapi.proxy = other.openapi.proxy
106
+ merged.repo.entries = self.repo.entries | other.repo.entries
64
107
  return merged
65
108
 
66
109
 
@@ -25,6 +25,7 @@ from apiscope.config import (
25
25
  get_project_config_path,
26
26
  )
27
27
  from apiscope.openapi import openapi_app
28
+ from apiscope.repo import check_repo_deps, repo_app
28
29
  from apiscope.rfc import check_rfc_deps, rfc_app
29
30
  from apiscope.schema import CommandContext
30
31
 
@@ -57,6 +58,7 @@ def callback(ctx: typer.Context) -> None:
57
58
 
58
59
  app.add_typer(openapi_app, name="openapi")
59
60
  app.add_typer(rfc_app, name="rfc")
61
+ app.add_typer(repo_app, name="repo")
60
62
 
61
63
  # ==============================================================================
62
64
  # commands
@@ -74,7 +76,7 @@ def health(
74
76
 
75
77
  # gather issues from all modules
76
78
  issues: list[str] = []
77
- for label, check in [("rfc", check_rfc_deps)]:
79
+ for label, check in [("rfc", check_rfc_deps), ("repo", check_repo_deps)]:
78
80
  err = check()
79
81
  if err is not None:
80
82
  issues.append(f"[{label}] {err}")
@@ -7,7 +7,7 @@ from typing import Any
7
7
  import typer
8
8
 
9
9
  from apiscope.config import CACHE_ROOT, Config
10
- from apiscope.openapi.fetch import fetch_openapi_spec
10
+ from apiscope.openapi.fetch import fetch_openapi_spec, openapi_cache_path
11
11
  from apiscope.openapi.reader import HttpMethod, OpenapiReader
12
12
  from apiscope.openapi.schema import OpenapiCommandContext
13
13
  from apiscope.openapi.spec import spec_app
@@ -26,16 +26,20 @@ def _resolve_source(alias_or_source: str, config: Config) -> str:
26
26
  return alias_or_source
27
27
 
28
28
 
29
- def _load_reader(source: str, cache_dir: Path, proxy: str | None = None) -> OpenapiReader:
30
- cached = fetch_openapi_spec(source, cache_dir, proxy)
29
+ def _load_reader(
30
+ source: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
31
+ ) -> OpenapiReader:
32
+ cached = fetch_openapi_spec(source, cache_dir, proxy, refresh=refresh)
31
33
  return OpenapiReader.load(cached)
32
34
 
33
35
 
34
- def _get_reader(source: str, ctx: typer.Context) -> OpenapiReader:
36
+ def _get_reader(source: str, ctx: typer.Context, force: bool = False) -> OpenapiReader:
35
37
  resolved = _resolve_source(source, ctx.obj.config)
36
38
  cache_dir = ctx.obj.openapi_command_context.cache_dir
37
39
  proxy = ctx.obj.config.openapi.proxy
38
- return _load_reader(resolved, cache_dir, proxy)
40
+ cache_path = openapi_cache_path(resolved, cache_dir)
41
+ stale = ctx.obj.config.openapi.is_stale_path(cache_path)
42
+ return _load_reader(resolved, cache_dir, proxy, refresh=force or stale)
39
43
 
40
44
 
41
45
  # ==============================================================================
@@ -61,9 +65,9 @@ def list_operations(
61
65
  source: str = typer.Argument(help="alias or path to the OpenAPI spec"),
62
66
  tag: str | None = typer.Option(default=None, help="filter by tag"),
63
67
  method: str | None = typer.Option(default=None, help="filter by HTTP method"),
68
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
64
69
  ) -> None:
65
- # load and resolve the spec
66
- reader = _get_reader(source, ctx)
70
+ reader = _get_reader(source, ctx, force=force)
67
71
 
68
72
  # collect all operations (path + method + identity fields)
69
73
  operations: list[dict[str, Any]] = []
@@ -106,9 +110,9 @@ def describe_operation(
106
110
  request: bool = typer.Option(
107
111
  default=False, show_default=False, help="show only request fields, omit responses"
108
112
  ),
113
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
109
114
  ) -> None:
110
- # load and resolve the spec
111
- reader = _get_reader(source, ctx)
115
+ reader = _get_reader(source, ctx, force=force)
112
116
 
113
117
  # merge path-item parameters with operation parameters
114
118
  path_item = reader.paths.get(path, {})
@@ -132,8 +136,9 @@ def describe_operation(
132
136
  def show_info(
133
137
  ctx: typer.Context,
134
138
  source: str = typer.Argument(help="alias or path to the OpenAPI spec"),
139
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache"),
135
140
  ) -> None:
136
- reader = _get_reader(source, ctx)
141
+ reader = _get_reader(source, ctx, force=force)
137
142
 
138
143
  META_KEYS = ("openapi", "info", "servers", "tags", "security", "externalDocs")
139
144
  meta = {k: v for k, v in reader.raw.items() if k in META_KEYS}
@@ -0,0 +1,58 @@
1
+ # apiscope/openapi/fetch.py
2
+
3
+ import shutil
4
+ from hashlib import sha256
5
+ from pathlib import Path
6
+ from urllib.parse import unquote, urlparse
7
+
8
+ import httpx
9
+
10
+ OPENAPI_EXTENSIONS = {".json", ".yaml", ".yml"}
11
+
12
+
13
+ def _cache_key(source: str) -> str:
14
+ return sha256(source.encode()).hexdigest()
15
+
16
+
17
+ def openapi_cache_path(source: str, cache_dir: Path) -> Path:
18
+ if "://" in source:
19
+ path = unquote(urlparse(source).path).rstrip()
20
+ else:
21
+ path = source
22
+ suffix = Path(path).suffix.lower()
23
+ ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
24
+ return cache_dir / f"{_cache_key(source)}{ext}"
25
+
26
+
27
+ def _fetch_local(source: str, cache_dir: Path, refresh: bool = False) -> Path:
28
+ cache_path = openapi_cache_path(source, cache_dir)
29
+
30
+ if refresh or not cache_path.exists():
31
+ src = Path(source).expanduser().resolve()
32
+ shutil.copy2(src, cache_path)
33
+
34
+ return cache_path
35
+
36
+
37
+ def _fetch_remote(
38
+ url: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
39
+ ) -> Path:
40
+ cache_path = openapi_cache_path(url, cache_dir)
41
+
42
+ if refresh or not cache_path.exists():
43
+ client_kwargs: dict = {}
44
+ if proxy is not None:
45
+ client_kwargs["proxy"] = proxy
46
+ resp = httpx.get(url, follow_redirects=True, **client_kwargs)
47
+ resp.raise_for_status()
48
+ cache_path.write_bytes(resp.content)
49
+
50
+ return cache_path
51
+
52
+
53
+ def fetch_openapi_spec(
54
+ source: str, cache_dir: Path, proxy: str | None = None, refresh: bool = False
55
+ ) -> Path:
56
+ if "://" in source:
57
+ return _fetch_remote(source, cache_dir, proxy, refresh=refresh)
58
+ return _fetch_local(source, cache_dir, refresh=refresh)
@@ -0,0 +1,6 @@
1
+ # apiscope/repo/__init__.py
2
+
3
+ from apiscope.repo.app import app as repo_app
4
+ from apiscope.repo.app import check_deps as check_repo_deps
5
+
6
+ __all__ = ["repo_app", "check_repo_deps"]
@@ -0,0 +1,152 @@
1
+ # apiscope/repo/app.py
2
+
3
+ import shutil
4
+ from pathlib import Path
5
+
6
+ import typer
7
+
8
+ from apiscope.config import (
9
+ CACHE_ROOT,
10
+ DEFAULT_CONFIG_PATH,
11
+ TMP_ROOT,
12
+ Config,
13
+ RepoEntryConfig,
14
+ get_project_config_path,
15
+ )
16
+ from apiscope.repo.fetch import sync_entry
17
+ from apiscope.repo.schema import RepoCommandContext, RepoEntry
18
+
19
+ app = typer.Typer(help="sync documentation from git repositories")
20
+
21
+
22
+ # ==============================================================================
23
+ # public helpers
24
+ # ==============================================================================
25
+
26
+
27
+ def check_deps() -> str | None:
28
+ if shutil.which("git") is None:
29
+ return "git is required but not found in PATH"
30
+ return None
31
+
32
+
33
+ # ==============================================================================
34
+ # helpers
35
+ # ==============================================================================
36
+
37
+
38
+ def _resolve_target(global_flag: bool) -> Path:
39
+ if global_flag:
40
+ return DEFAULT_CONFIG_PATH
41
+ project = get_project_config_path()
42
+ if project is not None and project.exists():
43
+ return project
44
+ return DEFAULT_CONFIG_PATH
45
+
46
+
47
+ # ==============================================================================
48
+ # callback
49
+ # ==============================================================================
50
+
51
+
52
+ @app.callback()
53
+ def repo_callback(
54
+ ctx: typer.Context,
55
+ global_flag: bool = typer.Option(
56
+ False, "--global", "-g", help="edit global config instead of project config"
57
+ ),
58
+ ) -> None:
59
+ # check required external tools
60
+ err = check_deps()
61
+ if err is not None:
62
+ typer.echo(err, err=True)
63
+ raise typer.Exit(code=1)
64
+
65
+ # prepare cache directories
66
+ repo_cache_dir = CACHE_ROOT / "repo"
67
+ repo_tmp_dir = TMP_ROOT
68
+ repo_cache_dir.mkdir(parents=True, exist_ok=True)
69
+ repo_tmp_dir.mkdir(parents=True, exist_ok=True)
70
+
71
+ # inject context for subcommands
72
+ ctx.obj.repo_command_context = RepoCommandContext(
73
+ cache_dir=repo_cache_dir,
74
+ tmp_dir=repo_tmp_dir,
75
+ global_flag=global_flag,
76
+ )
77
+
78
+
79
+ # ==============================================================================
80
+ # commands
81
+ # ==============================================================================
82
+
83
+
84
+ @app.command(name="add", help="register a repo for syncing")
85
+ def add_repo(
86
+ ctx: typer.Context,
87
+ url: str = typer.Argument(help="git clone URL"),
88
+ dir: str = typer.Argument(help="directory within repo to extract"),
89
+ target: str = typer.Option(
90
+ "branch:main", "--target", help="ref target (branch:, tag:, commit:)"
91
+ ),
92
+ ) -> None:
93
+ # duplicate check against merged entries
94
+ if url in ctx.obj.config.repo.entries:
95
+ typer.echo(f"<{url}> already registered", err=True)
96
+ raise typer.Exit(code=1)
97
+
98
+ target_path = _resolve_target(ctx.obj.repo_command_context.global_flag)
99
+ with Config.edit(target_path) as cfg:
100
+ cfg.repo.entries[url] = RepoEntryConfig(dir=dir, target=target)
101
+
102
+ typer.echo(f"registered <{url}>")
103
+
104
+
105
+ @app.command(name="remove", help="remove a registered repo")
106
+ def remove_repo(
107
+ ctx: typer.Context,
108
+ url: str = typer.Argument(help="git clone URL"),
109
+ ) -> None:
110
+ target_path = _resolve_target(ctx.obj.repo_command_context.global_flag)
111
+ with Config.edit(target_path) as cfg:
112
+ if url not in cfg.repo.entries:
113
+ typer.echo(f"<{url}> not registered", err=True)
114
+ raise typer.Exit(code=1)
115
+ del cfg.repo.entries[url]
116
+
117
+ typer.echo(f"removed <{url}>")
118
+
119
+
120
+ @app.command(name="list", help="list registered repos")
121
+ def list_repos(ctx: typer.Context) -> None:
122
+ entries = RepoEntry.from_config(ctx.obj.config.repo)
123
+ if not entries:
124
+ typer.echo("no repos registered")
125
+ return
126
+
127
+ cache_dir = ctx.obj.repo_command_context.cache_dir
128
+
129
+ for entry in entries:
130
+ doc_path = cache_dir / entry.sha256_id
131
+ typer.echo(f"- [{entry.url}] [{entry.target}@{entry.dir}] <{doc_path}>")
132
+
133
+
134
+ @app.command(name="sync", help="sync all registered repos to cache")
135
+ def sync_repos(
136
+ ctx: typer.Context,
137
+ force: bool = typer.Option(False, "--force", help="force re-sync ignoring cache TTL"),
138
+ ) -> None:
139
+ entries = RepoEntry.from_config(ctx.obj.config.repo)
140
+ if not entries:
141
+ typer.echo("no repos registered")
142
+ return
143
+
144
+ rc = ctx.obj.repo_command_context
145
+ ttl = ctx.obj.config.repo.cache_ttl
146
+
147
+ for entry in entries:
148
+ err = sync_entry(entry, rc.tmp_dir, rc.cache_dir, force=force, ttl=ttl)
149
+ if err is not None:
150
+ typer.echo(f"sync failed for <{entry.url}>: {err}", err=True)
151
+ else:
152
+ typer.echo(f"synced <{entry.url}>")
@@ -0,0 +1,132 @@
1
+ # apiscope/repo/fetch.py
2
+
3
+ import shutil
4
+ import subprocess
5
+ import time
6
+ from pathlib import Path
7
+
8
+ from apiscope.repo.schema import RepoEntry
9
+
10
+ # ==============================================================================
11
+ # git helpers
12
+ # ==============================================================================
13
+
14
+
15
+ def _run_git(args: list[str], cwd: Path | None = None) -> str | None:
16
+ result = subprocess.run(
17
+ ["git", *args],
18
+ capture_output=True,
19
+ text=True,
20
+ cwd=cwd,
21
+ )
22
+ if result.returncode != 0:
23
+ return result.stderr.strip()
24
+ return None
25
+
26
+
27
+ def _clone_repo(url: str, work_dir: Path, branch: str | None = None) -> str | None:
28
+ args = [
29
+ "clone",
30
+ "--depth=1",
31
+ "--filter=blob:none",
32
+ "--sparse",
33
+ ]
34
+ if branch is not None:
35
+ args.extend(["--branch", branch])
36
+ args.extend([url, str(work_dir)])
37
+ return _run_git(args)
38
+
39
+
40
+ def _fetch_commit(work_dir: Path, commit: str) -> str | None:
41
+ err = _run_git(["fetch", "origin", "--depth=1", commit], cwd=work_dir)
42
+ if err is not None:
43
+ return err
44
+ return _run_git(["checkout", "FETCH_HEAD"], cwd=work_dir)
45
+
46
+
47
+ def _sparse_set(work_dir: Path, directory: str) -> str | None:
48
+ return _run_git(["sparse-checkout", "set", directory], cwd=work_dir)
49
+
50
+
51
+ # ==============================================================================
52
+ # sync
53
+ # ==============================================================================
54
+
55
+
56
+ def _copy_docs(work_dir: Path, directory: str, dst: Path) -> str | None:
57
+ src = work_dir / directory
58
+ if not src.exists():
59
+ return f"directory [{directory}] not found in <{work_dir}>"
60
+ shutil.rmtree(dst, ignore_errors=True)
61
+ dst.mkdir(parents=True, exist_ok=True)
62
+ try:
63
+ shutil.copytree(src, dst, dirs_exist_ok=True)
64
+ except shutil.Error as e:
65
+ return str(e)
66
+ return None
67
+
68
+
69
+ def sync_entry(
70
+ entry: RepoEntry,
71
+ tmp_dir: Path,
72
+ cache_dir: Path,
73
+ *,
74
+ force: bool = False,
75
+ ttl: int,
76
+ ) -> str | None:
77
+ work_dir = tmp_dir / entry.sha256_id
78
+ dst_dir = cache_dir / entry.sha256_id
79
+
80
+ # skip if cache is fresh (not stale) and not forced
81
+ if not force and not _is_stale(dst_dir, ttl):
82
+ return None
83
+
84
+ # cleanup previous state
85
+ shutil.rmtree(work_dir, ignore_errors=True)
86
+ shutil.rmtree(dst_dir, ignore_errors=True)
87
+
88
+ ref = entry.reference
89
+
90
+ if ref is not None and ref[0] == "commit":
91
+ # commit ref: clone default branch first, then fetch + checkout specific commit
92
+ ref_name = ref[1]
93
+ err = _clone_repo(entry.url, work_dir)
94
+ if err is not None:
95
+ return err
96
+ err = _fetch_commit(work_dir, ref_name)
97
+ if err is not None:
98
+ return err
99
+ else:
100
+ # branch or tag: --branch accepts both (tags checkout in detached HEAD)
101
+ branch = ref[1] if ref is not None else "main"
102
+ err = _clone_repo(entry.url, work_dir, branch=branch)
103
+ if err is not None:
104
+ return err
105
+
106
+ # sparse checkout target directory
107
+ err = _sparse_set(work_dir, entry.dir)
108
+ if err is not None:
109
+ return err
110
+
111
+ # copy extracted docs to cache
112
+ err = _copy_docs(work_dir, entry.dir, dst_dir)
113
+ if err is not None:
114
+ return err
115
+
116
+ # cleanup working directory
117
+ shutil.rmtree(work_dir, ignore_errors=True)
118
+
119
+ return None
120
+
121
+
122
+ # ==============================================================================
123
+ # freshness
124
+ # ==============================================================================
125
+
126
+
127
+ def _is_stale(path: Path, ttl: int) -> bool:
128
+ if not path.exists():
129
+ return True
130
+ stat = path.stat()
131
+ latest = max(stat.st_mtime, stat.st_ctime)
132
+ return time.time() - latest > ttl
@@ -0,0 +1,53 @@
1
+ # apiscope/repo/schema.py
2
+
3
+ import hashlib
4
+ from pathlib import Path
5
+ from typing import TYPE_CHECKING
6
+
7
+ from pydantic import BaseModel
8
+
9
+ if TYPE_CHECKING:
10
+ from apiscope.config import RepoConfig
11
+
12
+
13
+ class RepoEntry(BaseModel):
14
+ url: str
15
+ dir: str
16
+ target: str
17
+
18
+ def __hash__(self) -> int:
19
+ return hash(self.sha256_id)
20
+
21
+ @property
22
+ def sha256_id(self) -> str:
23
+ # first 8 chars of sha256 hex digest; sufficient for collision resistance
24
+ return hashlib.sha256(self.url.encode()).hexdigest()[:8]
25
+
26
+ @property
27
+ def reference(self) -> tuple[str, str] | None:
28
+ if self.target.startswith("branch:"):
29
+ return "branch", self.target[len("branch:") :]
30
+ elif self.target.startswith("tag:"):
31
+ return "tag", self.target[len("tag:") :]
32
+ elif self.target.startswith("commit:"):
33
+ return "commit", self.target[len("commit:") :]
34
+ else:
35
+ return None
36
+
37
+ @classmethod
38
+ def from_config(cls, config: "RepoConfig") -> set["RepoEntry"]:
39
+ return {cls(url=url, dir=ec.dir, target=ec.target) for url, ec in config.entries.items()}
40
+
41
+
42
+ # ==============================================================================
43
+ # context
44
+ # ==============================================================================
45
+
46
+
47
+ class RepoCommandContext(BaseModel):
48
+ # computed properties
49
+ cache_dir: Path
50
+ tmp_dir: Path
51
+
52
+ # sub command options
53
+ global_flag: bool
@@ -88,10 +88,20 @@ def _show_page_info(content: str, number: int, rfc_ctx: RfcCommandContext) -> No
88
88
 
89
89
 
90
90
  @app.command(name="sync", help="download RFC metadata index via rsync")
91
- def sync_index(ctx: typer.Context) -> None:
91
+ def sync_index(
92
+ ctx: typer.Context,
93
+ force: bool = typer.Option(False, "--force", help="force re-sync ignoring cache TTL"),
94
+ ) -> None:
92
95
  rfc_ctx = ctx.obj.rfc_command_context
96
+ rfc_config = ctx.obj.config.rfc
97
+ index_dir = rfc_ctx.index_json_dir
98
+
99
+ if not force and not rfc_config.is_stale_path(index_dir):
100
+ typer.echo("index is up to date")
101
+ return
102
+
93
103
  typer.echo("syncing via rsync...")
94
- err = fetch_all_index_json(rfc_ctx.index_json_dir)
104
+ err = fetch_all_index_json(index_dir)
95
105
  if err is not None:
96
106
  typer.echo(f"sync failed\n{err}", err=True)
97
107
  raise typer.Exit(code=1)
@@ -139,8 +149,10 @@ def read_content(
139
149
  section: str | None = typer.Option(None, "--section", help="extract by section id (XML only)"),
140
150
  page: int | None = typer.Option(None, "--page", help="extract by page number (TXT only)"),
141
151
  json_output: bool = typer.Option(False, "--json", help="output as JSON"),
152
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache TTL"),
142
153
  ) -> None:
143
154
  rfc_ctx = ctx.obj.rfc_command_context
155
+ rfc_config = ctx.obj.config.rfc
144
156
  meta = rfc_ctx.get_index_json(number)
145
157
  if meta is None:
146
158
  typer.echo(f"rfc {number} not found", err=True)
@@ -158,7 +170,7 @@ def read_content(
158
170
  raise typer.Exit(code=1)
159
171
 
160
172
  # ensure content is available locally
161
- if not content_path.exists():
173
+ if force or rfc_config.is_stale_path(content_path):
162
174
  typer.echo(f"fetching <{content_path.name}> via rsync...")
163
175
  err = fetch_content_by_number(number, content_path.parent, fmt)
164
176
  if err is not None:
@@ -235,8 +247,10 @@ def search_rfc(
235
247
  no_snippet: bool = typer.Option(False, "--no-snippet", help="hide content snippet"),
236
248
  limit: int = typer.Option(20, "--limit", help="max results shown"),
237
249
  offset: int = typer.Option(0, "--offset", help="skip first N results"),
250
+ force: bool = typer.Option(False, "--force", help="force re-fetch ignoring cache TTL"),
238
251
  ) -> None:
239
252
  rfc_ctx = ctx.obj.rfc_command_context
253
+ rfc_config = ctx.obj.config.rfc
240
254
  # step 1: dispatch mode
241
255
  index_filters = [status, since, until, author, title, abstract, keywords_field, source]
242
256
  fulltext_mode = number is not None and term is not None
@@ -272,7 +286,7 @@ def search_rfc(
272
286
  raise typer.Exit(code=1)
273
287
 
274
288
  content_path = rfc_ctx.get_content_txt_path(number)
275
- if not content_path.exists():
289
+ if force or rfc_config.is_stale_path(content_path):
276
290
  typer.echo(f"fetching <{content_path.name}> via rsync...")
277
291
  err = fetch_content_by_number(number, content_path.parent, "txt")
278
292
  if err is not None:
@@ -4,6 +4,7 @@ from pydantic import BaseModel
4
4
 
5
5
  from apiscope.config import Config
6
6
  from apiscope.openapi.schema import OpenapiCommandContext
7
+ from apiscope.repo.schema import RepoCommandContext
7
8
  from apiscope.rfc.schema import RfcCommandContext
8
9
 
9
10
 
@@ -11,3 +12,4 @@ class CommandContext(BaseModel):
11
12
  config: Config
12
13
  openapi_command_context: OpenapiCommandContext | None = None
13
14
  rfc_command_context: RfcCommandContext | None = None
15
+ repo_command_context: RepoCommandContext | None = None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-api-scope
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: read and cache structured documents from remote for LLM agents
5
5
  Author-email: D7x7z49 <85430783+D7x7z49@users.noreply.github.com>
6
6
  License: MIT License
@@ -73,9 +73,20 @@ browse OpenAPI specifications with subcommands for discovering, listing, and des
73
73
 
74
74
  aliases let you register frequently used specs once and reference them by short name. fetching is transparent — local copies are cached for fast repeat access, and a proxy can be configured for restricted networks.
75
75
 
76
+ ### rfc
77
+
78
+ read, search, and navigate RFC documents from the IETF.
79
+
80
+ the metadata index is mirrored once via rsync. individual text files are fetched on demand and cached locally. you can browse the table of contents, jump to a section (XML) or page (TXT), filter the index by status or source, and run keyword searches against fulltext content.
81
+
82
+ ### repo
83
+
84
+ sync documentation directories from any git repository to a local cache.
85
+
86
+ register repositories by URL with a target subdirectory and a ref — a branch, tag, or commit. the sync command clones with shallow depth, blobless filter, and sparse checkout so only the needed tree and files come over the wire. extracted docs land in the cache under a stable hash path, and a configurable TTL avoids redundant re-fetches. provider-agnostic, zero auth.
87
+
76
88
  ## future
77
89
 
78
- - read RFC documents by number
79
90
  - read academic papers from arxiv
80
91
  - more formal document formats as the need arises
81
92
 
@@ -13,6 +13,10 @@ apiscope/openapi/schema.py
13
13
  apiscope/openapi/spec/__init__.py
14
14
  apiscope/openapi/spec/app.py
15
15
  apiscope/openapi/spec/schema.py
16
+ apiscope/repo/__init__.py
17
+ apiscope/repo/app.py
18
+ apiscope/repo/fetch.py
19
+ apiscope/repo/schema.py
16
20
  apiscope/rfc/__init__.py
17
21
  apiscope/rfc/app.py
18
22
  apiscope/rfc/fetch.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "llm-api-scope"
7
- version = "0.5.0"
7
+ version = "0.6.0"
8
8
  description = "read and cache structured documents from remote for LLM agents"
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
@@ -1,49 +0,0 @@
1
- # apiscope/openapi/fetch.py
2
-
3
- import shutil
4
- from hashlib import sha256
5
- from pathlib import Path
6
- from urllib.parse import unquote, urlparse
7
-
8
- import httpx
9
-
10
- OPENAPI_EXTENSIONS = {".json", ".yaml", ".yml"}
11
-
12
-
13
- def _cache_key(source: str) -> str:
14
- return sha256(source.encode()).hexdigest()
15
-
16
-
17
- def _fetch_local(source: str, cache_dir: Path) -> Path:
18
- suffix = Path(source).suffix.lower()
19
- ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
20
- cache_path = cache_dir / f"{_cache_key(source)}{ext}"
21
-
22
- if not cache_path.exists():
23
- src = Path(source).expanduser().resolve()
24
- shutil.copy2(src, cache_path)
25
-
26
- return cache_path
27
-
28
-
29
- def _fetch_remote(url: str, cache_dir: Path, proxy: str | None = None) -> Path:
30
- path = unquote(urlparse(url).path).rstrip()
31
- suffix = Path(path).suffix.lower()
32
- ext = suffix if suffix in OPENAPI_EXTENSIONS else ".json"
33
- cache_path = cache_dir / f"{_cache_key(url)}{ext}"
34
-
35
- if not cache_path.exists():
36
- client_kwargs: dict = {}
37
- if proxy is not None:
38
- client_kwargs["proxy"] = proxy
39
- resp = httpx.get(url, follow_redirects=True, **client_kwargs)
40
- resp.raise_for_status()
41
- cache_path.write_bytes(resp.content)
42
-
43
- return cache_path
44
-
45
-
46
- def fetch_openapi_spec(source: str, cache_dir: Path, proxy: str | None = None) -> Path:
47
- if "://" in source:
48
- return _fetch_remote(source, cache_dir, proxy)
49
- return _fetch_local(source, cache_dir)
File without changes
File without changes