java-codebase-rag 0.9.7__py3-none-any.whl → 0.10.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/analysis/pr_analysis.py +33 -3
- java_codebase_rag/ast/ast_java.py +2 -1
- java_codebase_rag/cli.py +23 -8
- java_codebase_rag/config.py +68 -1
- java_codebase_rag/graph/build_ast_graph.py +123 -4
- java_codebase_rag/graph/graph_types.py +109 -22
- java_codebase_rag/graph/ladybug_queries.py +45 -2
- java_codebase_rag/index/java_index_flow_lancedb.py +10 -16
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +3 -1
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +3 -1
- java_codebase_rag/jrag.py +627 -661
- java_codebase_rag/jrag_render.py +160 -3
- java_codebase_rag/lance_optimize.py +11 -12
- java_codebase_rag/mcp/mcp_v2.py +2 -1
- java_codebase_rag/pipeline.py +47 -1
- java_codebase_rag/read_payloads.py +781 -0
- java_codebase_rag/search/search_lancedb.py +138 -6
- java_codebase_rag/search/search_lexical.py +140 -30
- java_codebase_rag/search/search_scoring.py +108 -0
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +230 -0
- java_codebase_rag/watch/daemon.py +368 -0
- java_codebase_rag/watch/lock.py +201 -0
- java_codebase_rag/watch/paths.py +76 -0
- java_codebase_rag/watch/protocol.py +122 -0
- java_codebase_rag/watch/server.py +273 -0
- java_codebase_rag/watch/warm.py +105 -0
- java_codebase_rag/watch/watcher.py +352 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/METADATA +30 -31
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/RECORD +34 -24
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/licenses/LICENSE +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/top_level.txt +0 -0
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""IPC client for the ``jrag watch`` daemon: try-the-daemon, cold-fall-back.
|
|
2
|
+
|
|
3
|
+
The cold read path is BYTE-IDENTICAL to today: when no daemon is alive (the
|
|
4
|
+
common case — including every ``jrag`` subprocess invocation that is not a
|
|
5
|
+
``jrag watch`` process), :func:`get_payload` falls back to the same
|
|
6
|
+
``<cmd>_payload`` core the handler already calls, loading the graph via
|
|
7
|
+
``_load_graph(cfg)`` (a cache hit on the already-loaded singleton). The daemon
|
|
8
|
+
is a pure accelerator: it never changes observable output, only latency.
|
|
9
|
+
|
|
10
|
+
Contract (pinned by task-8 brief):
|
|
11
|
+
|
|
12
|
+
* :func:`is_daemon_alive` -> ``socket_path`` exists AND
|
|
13
|
+
:meth:`ProjectLock.read_holder` returns a live pid.
|
|
14
|
+
* :func:`request` -> ``response.result`` dict; raises :class:`DaemonUnavailable`
|
|
15
|
+
(no daemon / version mismatch / hung) or :class:`DaemonError` (``ok=False``).
|
|
16
|
+
* :func:`get_payload` -> the single seam each read handler calls. Tries the
|
|
17
|
+
daemon; on :class:`DaemonUnavailable`/:class:`DaemonError` runs the cold
|
|
18
|
+
``cold_core(argparse.Namespace(**args), cfg, _load_graph(cfg))``. On success it
|
|
19
|
+
RECONSTRUCTS the payload object from the daemon's serialized dict so the
|
|
20
|
+
handler's downstream project+render runs unchanged.
|
|
21
|
+
|
|
22
|
+
Reconstruction is the lossless inverse of :func:`watch.server.serialize`:
|
|
23
|
+
pydantic ``*Output`` models round-trip via ``model_dump(mode="json")`` /
|
|
24
|
+
``model_validate``; the traversal payloads (callers/callees/flow) are plain
|
|
25
|
+
dicts and pass through. The one non-pydantic case is ``find`` query mode, whose
|
|
26
|
+
``rows`` are :class:`SymbolHit` dataclasses accessed by attribute in the renderer
|
|
27
|
+
(``row.id``); those are rebuilt via ``SymbolHit(**row)`` so the renderer is
|
|
28
|
+
untouched. (The task-8 brief literal said "pass-through" for query mode; that
|
|
29
|
+
would break the renderer's attribute access on the hot path, so the rows are
|
|
30
|
+
rebuilt instead — see task-8 report.)
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import socket
|
|
35
|
+
from typing import TYPE_CHECKING, Any, Callable
|
|
36
|
+
|
|
37
|
+
from java_codebase_rag.jrag import _load_graph
|
|
38
|
+
from java_codebase_rag.watch.lock import ProjectLock
|
|
39
|
+
from java_codebase_rag.watch.paths import socket_path
|
|
40
|
+
from java_codebase_rag.watch.protocol import (
|
|
41
|
+
PROTOCOL_VERSION,
|
|
42
|
+
ProtocolMismatch,
|
|
43
|
+
Request,
|
|
44
|
+
decode_response,
|
|
45
|
+
encode_request,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
if TYPE_CHECKING:
|
|
49
|
+
from pathlib import Path
|
|
50
|
+
|
|
51
|
+
# Connect/read budget so a hung daemon falls back to cold rather than blocking
|
|
52
|
+
# the read command. 2s is generous for a local AF_UNIX round trip (the daemon
|
|
53
|
+
# answers inline; a healthy response is sub-millisecond).
|
|
54
|
+
_DAEMON_TIMEOUT = 2.0
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class DaemonUnavailable(Exception):
|
|
58
|
+
"""No live daemon, an unreachable/hung daemon, or a protocol-version mismatch.
|
|
59
|
+
|
|
60
|
+
Always triggers the cold fallback in :func:`get_payload`.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class DaemonError(Exception):
|
|
65
|
+
"""The daemon answered ``ok=False`` (e.g. ``backend_error`` / ``stale_index``).
|
|
66
|
+
|
|
67
|
+
Attributes:
|
|
68
|
+
kind: the ``Response.error.kind`` string (``"backend_error"`` etc.).
|
|
69
|
+
message: the ``Response.error.message`` string.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
def __init__(self, kind: str, message: str) -> None:
|
|
73
|
+
self.kind = kind
|
|
74
|
+
self.message = message
|
|
75
|
+
super().__init__(f"{kind}: {message}")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def is_daemon_alive(index_dir: "Path") -> bool:
|
|
79
|
+
"""True iff the daemon's socket exists AND a live holder pid is recorded.
|
|
80
|
+
|
|
81
|
+
The pid check (:meth:`ProjectLock.read_holder`) returns ``None`` for a
|
|
82
|
+
missing/empty pid file or a dead/stale pid, so a leftover socket alone does
|
|
83
|
+
not count as "alive" — a crashed daemon's socket is ignored and the read
|
|
84
|
+
falls back to cold.
|
|
85
|
+
"""
|
|
86
|
+
if not socket_path(index_dir).exists():
|
|
87
|
+
return False
|
|
88
|
+
return ProjectLock.read_holder(index_dir) is not None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def request(index_dir: "Path", cmd: str, args: dict) -> dict:
|
|
92
|
+
"""Send one command to the daemon and return its ``result`` payload dict.
|
|
93
|
+
|
|
94
|
+
Raises:
|
|
95
|
+
DaemonUnavailable: not alive, unreachable/hung (timeout), closed
|
|
96
|
+
mid-response, or speaking a different protocol version (an older
|
|
97
|
+
daemon). All of these trigger the cold fallback.
|
|
98
|
+
DaemonError: the daemon responded ``ok=False``.
|
|
99
|
+
"""
|
|
100
|
+
if not is_daemon_alive(index_dir):
|
|
101
|
+
raise DaemonUnavailable(f"no live daemon for {index_dir}")
|
|
102
|
+
|
|
103
|
+
sock_path = socket_path(index_dir)
|
|
104
|
+
try:
|
|
105
|
+
with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as sock:
|
|
106
|
+
sock.settimeout(_DAEMON_TIMEOUT)
|
|
107
|
+
sock.connect(str(sock_path))
|
|
108
|
+
sock.sendall(encode_request(Request(v=PROTOCOL_VERSION, cmd=cmd, args=args)))
|
|
109
|
+
line = _readline(sock)
|
|
110
|
+
except OSError as exc:
|
|
111
|
+
# Timeout, connection refused, etc. -> cold fallback (never block).
|
|
112
|
+
raise DaemonUnavailable(f"daemon unreachable at {sock_path}: {exc}") from exc
|
|
113
|
+
|
|
114
|
+
try:
|
|
115
|
+
response = decode_response(line)
|
|
116
|
+
except (ProtocolMismatch, ValueError) as exc:
|
|
117
|
+
# Version mismatch (older daemon) or a blank/partial line (daemon gone)
|
|
118
|
+
# -> cold fallback rather than crashing the read command.
|
|
119
|
+
raise DaemonUnavailable(f"daemon response undecodable: {exc}") from exc
|
|
120
|
+
|
|
121
|
+
if not response.ok:
|
|
122
|
+
err = response.error
|
|
123
|
+
raise DaemonError(
|
|
124
|
+
err.kind if err is not None else "unknown",
|
|
125
|
+
err.message if err is not None else "",
|
|
126
|
+
)
|
|
127
|
+
return response.result
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def get_payload(cmd: str, args: dict, cfg, *, cold_core: Callable[..., Any]) -> Any:
|
|
131
|
+
"""Return the payload for ``cmd``, trying the daemon first then cold-falling-back.
|
|
132
|
+
|
|
133
|
+
``args`` is the command's full parsed-namespace dict (``vars(args)``);
|
|
134
|
+
``cold_core`` is the matching ``<cmd>_payload(args, cfg, graph)`` core.
|
|
135
|
+
|
|
136
|
+
* Daemon success: the serialized payload dict is RECONSTRUCTED into the same
|
|
137
|
+
object the cold core returns, so the handler's render path is unchanged.
|
|
138
|
+
* DaemonUnavailable / DaemonError: the cold core runs against
|
|
139
|
+
``_load_graph(cfg)`` — byte-identical to today (the daemon is absent in
|
|
140
|
+
every non-``watch`` invocation, and an ``ok=False`` frame re-surfaces as
|
|
141
|
+
the cold core's own ``PayloadError`` → identical error envelope + rc).
|
|
142
|
+
"""
|
|
143
|
+
try:
|
|
144
|
+
result = request(cfg.index_dir, cmd, args)
|
|
145
|
+
except (DaemonUnavailable, DaemonError):
|
|
146
|
+
return cold_core(argparse_namespace(args), cfg, _load_graph(cfg))
|
|
147
|
+
return _reconstruct(cmd, result)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# ---------------------------------------------------------------------------
|
|
151
|
+
# internals
|
|
152
|
+
# ---------------------------------------------------------------------------
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _readline(sock: socket.socket) -> bytes:
|
|
156
|
+
"""Read until newline (the response is one NDJSON line). Returns whatever was
|
|
157
|
+
received before newline/EOF; the caller decodes (and maps blanks to
|
|
158
|
+
``DaemonUnavailable``)."""
|
|
159
|
+
buf = b""
|
|
160
|
+
while b"\n" not in buf:
|
|
161
|
+
chunk = sock.recv(4096)
|
|
162
|
+
if not chunk:
|
|
163
|
+
break
|
|
164
|
+
buf += chunk
|
|
165
|
+
return buf
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _reconstruct(cmd: str, result: Any) -> Any:
|
|
169
|
+
"""Rebuild the payload object the cold core returns from the daemon's JSON dict.
|
|
170
|
+
|
|
171
|
+
Inverse of :func:`watch.server.serialize`. See module docstring for the
|
|
172
|
+
find query-mode ``SymbolHit`` rebuild rationale.
|
|
173
|
+
"""
|
|
174
|
+
if cmd == "search":
|
|
175
|
+
from java_codebase_rag.mcp.mcp_v2 import SearchOutput
|
|
176
|
+
|
|
177
|
+
return SearchOutput.model_validate(result)
|
|
178
|
+
|
|
179
|
+
if cmd == "inspect":
|
|
180
|
+
from java_codebase_rag.mcp.mcp_v2 import DescribeOutput
|
|
181
|
+
|
|
182
|
+
return {
|
|
183
|
+
"describe": DescribeOutput.model_validate(result["describe"]),
|
|
184
|
+
"node_id": result["node_id"],
|
|
185
|
+
"node_fqn": result["node_fqn"],
|
|
186
|
+
"file_location": result["file_location"],
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
if cmd == "find":
|
|
190
|
+
# The find payload is always a {"mode": "query"|"filter", ...} dict.
|
|
191
|
+
mode = result.get("mode")
|
|
192
|
+
if mode == "filter":
|
|
193
|
+
from java_codebase_rag.mcp.mcp_v2 import FindOutput
|
|
194
|
+
|
|
195
|
+
return {
|
|
196
|
+
"mode": "filter",
|
|
197
|
+
"kind": result["kind"],
|
|
198
|
+
"out": FindOutput.model_validate(result["out"]),
|
|
199
|
+
"limit": result["limit"],
|
|
200
|
+
}
|
|
201
|
+
# query mode: rows are SymbolHit dataclasses (attribute access in the
|
|
202
|
+
# renderer), rebuilt here so rendering is byte-identical.
|
|
203
|
+
from java_codebase_rag.graph.ladybug_queries import SymbolHit
|
|
204
|
+
|
|
205
|
+
return {
|
|
206
|
+
"mode": "query",
|
|
207
|
+
"rows": [SymbolHit(**row) for row in result["rows"]],
|
|
208
|
+
"raw_truncated": result["raw_truncated"],
|
|
209
|
+
"post_filter_active": result["post_filter_active"],
|
|
210
|
+
"limit": result["limit"],
|
|
211
|
+
"query": result["query"],
|
|
212
|
+
"kinds": result["kinds"],
|
|
213
|
+
"matched_mode": result["matched_mode"],
|
|
214
|
+
"identifier_matched": result["identifier_matched"],
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
# callers / callees / flow: plain dicts already (node/edge values are dicts
|
|
218
|
+
# of JSON-native scalars); pass through unchanged.
|
|
219
|
+
return result
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def argparse_namespace(args: dict):
|
|
223
|
+
"""Build an ``argparse.Namespace`` from the args dict (deferred import).
|
|
224
|
+
|
|
225
|
+
Imported lazily so this module's top level stays free of the ``argparse``
|
|
226
|
+
dependency needed only on the cold path; also a single seam for the cold
|
|
227
|
+
core's ``args`` reconstruction."""
|
|
228
|
+
import argparse
|
|
229
|
+
|
|
230
|
+
return argparse.Namespace(**args)
|
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
"""The ``jrag watch`` daemon process: assemble the watch components and run them.
|
|
2
|
+
|
|
3
|
+
``WatchDaemon`` is the capstone that wires together the building blocks built in
|
|
4
|
+
Tasks 2-10:
|
|
5
|
+
|
|
6
|
+
* :class:`watch.lock.ProjectLock` — single-writer mutual exclusion per project.
|
|
7
|
+
* :class:`watch.warm.WarmResources` — the warm embedding model + the read-only
|
|
8
|
+
graph reader (and the graph copy-on-write snapshot lifecycle).
|
|
9
|
+
* :class:`watch.server.WatchServer` — the AF_UNIX socket server that dispatches
|
|
10
|
+
read commands to the payload cores and ships the serialized payload.
|
|
11
|
+
* :class:`watch.watcher.SourceWatcher` — the file watcher + debounced per-type
|
|
12
|
+
reindex dispatcher.
|
|
13
|
+
|
|
14
|
+
Lifecycle (``run_foreground``):
|
|
15
|
+
|
|
16
|
+
1. Acquire the project lock. Held elsewhere -> stderr line + ``return 2``.
|
|
17
|
+
Unsupported platform -> stderr line + ``return 2``.
|
|
18
|
+
2. EAGERLY warm the embedding model so a load failure fails fast (stderr +
|
|
19
|
+
``on_event("error", …)`` + lock release + ``return 2``).
|
|
20
|
+
3. Install SIGINT/SIGTERM handlers that set a stop flag.
|
|
21
|
+
4. ``server.start()`` then ``watcher.start()``.
|
|
22
|
+
5. Write the state file (``paths.state_path``) so ``--status``/``--stop`` from
|
|
23
|
+
another process can see current truth.
|
|
24
|
+
6. Render a ``rich`` Live status panel (watcher state, last reindex) and
|
|
25
|
+
block on a wait loop until the stop flag is set.
|
|
26
|
+
7. Tear down in order — ``watcher.stop()`` → ``server.shutdown()`` →
|
|
27
|
+
``lock.release()`` → unlink socket + state file — and terminate with
|
|
28
|
+
``os._exit(0)``. The explicit ``os._exit`` (mirroring
|
|
29
|
+
``jrag._console_script_main``) skips interpreter finalization, which dodges
|
|
30
|
+
a racy pyarrow/lance worker-thread SIGABRT once the daemon has served a
|
|
31
|
+
``search`` (the read path loads lancedb in-process). ``run_foreground``
|
|
32
|
+
therefore NEVER returns normally on the serving path.
|
|
33
|
+
"""
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import json
|
|
37
|
+
import logging
|
|
38
|
+
import os
|
|
39
|
+
import signal
|
|
40
|
+
import sys
|
|
41
|
+
import threading
|
|
42
|
+
import time
|
|
43
|
+
from typing import TYPE_CHECKING, Any
|
|
44
|
+
|
|
45
|
+
from java_codebase_rag.watch import paths
|
|
46
|
+
from java_codebase_rag.watch.lock import (
|
|
47
|
+
LockHeldError,
|
|
48
|
+
ProjectLock,
|
|
49
|
+
WatchUnsupportedPlatform,
|
|
50
|
+
)
|
|
51
|
+
from java_codebase_rag.watch.server import WatchServer
|
|
52
|
+
from java_codebase_rag.watch.warm import WarmResources
|
|
53
|
+
from java_codebase_rag.watch.watcher import SourceWatcher
|
|
54
|
+
|
|
55
|
+
if TYPE_CHECKING:
|
|
56
|
+
from java_codebase_rag.config import ResolvedOperatorConfig
|
|
57
|
+
|
|
58
|
+
# State-file rewrites are throttled so a busy reindex burst does not hammer disk.
|
|
59
|
+
# The initial write (``force=True``) and the last reindex both go through
|
|
60
|
+
# immediately; ``--status`` readers tolerate a slightly-stale ``last_reindex``.
|
|
61
|
+
_STATE_WRITE_MIN_INTERVAL_S = 1.0
|
|
62
|
+
# The blocking loop's tick: how often the Live panel refreshes and how quickly a
|
|
63
|
+
# stop signal is observed. 0.5 s is responsive without burning CPU.
|
|
64
|
+
_LOOP_TICK_S = 0.5
|
|
65
|
+
|
|
66
|
+
log = logging.getLogger(__name__)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class WatchDaemon:
|
|
70
|
+
"""Assemble the watch components and serve until interrupted.
|
|
71
|
+
|
|
72
|
+
The daemon process holds the project lock for its entire lifetime; the state
|
|
73
|
+
file is the cross-process truth that ``--status``/``--stop`` read.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
def __init__(self, cfg: "ResolvedOperatorConfig") -> None:
|
|
77
|
+
self.cfg = cfg
|
|
78
|
+
self.lock = ProjectLock(cfg.index_dir)
|
|
79
|
+
self.warm = WarmResources(cfg)
|
|
80
|
+
self.server = WatchServer(self.warm, cfg)
|
|
81
|
+
self.watcher = SourceWatcher(
|
|
82
|
+
cfg,
|
|
83
|
+
self.warm,
|
|
84
|
+
debounce_ms=cfg.watch_debounce_ms,
|
|
85
|
+
backend=cfg.watch_backend,
|
|
86
|
+
poll_interval_ms=cfg.watch_poll_interval_ms,
|
|
87
|
+
on_event=self._record,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
self._stop = threading.Event()
|
|
91
|
+
self._state_lock = threading.Lock()
|
|
92
|
+
self._last_state_write = 0.0
|
|
93
|
+
self._state: dict[str, Any] = {
|
|
94
|
+
"started_at": None,
|
|
95
|
+
"pid": None,
|
|
96
|
+
"socket": str(paths.socket_path(cfg.index_dir)),
|
|
97
|
+
"last_reindex_at": None,
|
|
98
|
+
"last_reindex_kind": None,
|
|
99
|
+
"reindex_count": 0,
|
|
100
|
+
# ``queries_served`` is left at 0 for v1: wiring a query-count
|
|
101
|
+
# callback out of ``WatchServer`` (Task 7, approved) is out of scope
|
|
102
|
+
# for this task's commit surface and the brief marks it optional.
|
|
103
|
+
"queries_served": 0,
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
# ------------------------------------------------------------------
|
|
107
|
+
# public lifecycle
|
|
108
|
+
# ------------------------------------------------------------------
|
|
109
|
+
|
|
110
|
+
def run_foreground(self) -> int:
|
|
111
|
+
"""Serve until SIGINT/SIGTERM, then tear down and ``os._exit(0)``.
|
|
112
|
+
|
|
113
|
+
Early failure paths (lock held, unsupported platform, model-load error)
|
|
114
|
+
return ``2`` normally — they occur before the server accepts connections,
|
|
115
|
+
so no lance worker threads exist and interpreter finalization is safe.
|
|
116
|
+
|
|
117
|
+
The serving path (after ``server.start()``) terminates ONLY via
|
|
118
|
+
``os._exit(0)`` in :meth:`_shutdown`; the trailing ``return 0`` is
|
|
119
|
+
unreachable and exists to satisfy the ``-> int`` contract.
|
|
120
|
+
"""
|
|
121
|
+
# 1. Acquire the project lock (single writer per project).
|
|
122
|
+
try:
|
|
123
|
+
self.lock.acquire()
|
|
124
|
+
except LockHeldError as exc:
|
|
125
|
+
print(f"jrag watch: index in use by PID {exc.pid}", file=sys.stderr)
|
|
126
|
+
return 2
|
|
127
|
+
except WatchUnsupportedPlatform:
|
|
128
|
+
print("jrag watch: watch mode requires macOS/Linux", file=sys.stderr)
|
|
129
|
+
return 2
|
|
130
|
+
|
|
131
|
+
# 2. Eagerly warm the model so a load failure fails fast (before the
|
|
132
|
+
# server accepts a single query). The model is the only heavy,
|
|
133
|
+
# failure-prone resource that is not lazy on the read path.
|
|
134
|
+
try:
|
|
135
|
+
self.warm.model()
|
|
136
|
+
except Exception as exc: # noqa: BLE001 — report any load failure, then bail
|
|
137
|
+
print(f"jrag watch: failed to load embedding model: {exc}", file=sys.stderr)
|
|
138
|
+
self._record("error", {"phase": "model_load", "error": repr(exc)})
|
|
139
|
+
self.lock.release()
|
|
140
|
+
return 2
|
|
141
|
+
|
|
142
|
+
# 3. Install stop-signal handlers (main thread only).
|
|
143
|
+
signal.signal(signal.SIGINT, self._on_signal)
|
|
144
|
+
signal.signal(signal.SIGTERM, self._on_signal)
|
|
145
|
+
|
|
146
|
+
# 4a. Server start: bind the socket before starting the watcher so a
|
|
147
|
+
# query the moment the panel renders is already servable.
|
|
148
|
+
#
|
|
149
|
+
# We hold the EXCLUSIVE project lock, so we are the unique legitimate
|
|
150
|
+
# owner of this socket path: any pre-existing socket file is a corpse
|
|
151
|
+
# from a crashed prior daemon and MUST be cleared before bind, else
|
|
152
|
+
# AF_UNIX bind() fails with EADDRINUSE. ``server.start`` also defends
|
|
153
|
+
# against a stale socket for callers that do NOT hold the lock, but its
|
|
154
|
+
# guard (``read_holder is None``) is inert here precisely because we
|
|
155
|
+
# hold the lock — so the daemon clears its own stale socket itself.
|
|
156
|
+
stale_sock = paths.socket_path(self.cfg.index_dir)
|
|
157
|
+
try:
|
|
158
|
+
stale_sock.unlink()
|
|
159
|
+
except FileNotFoundError:
|
|
160
|
+
pass
|
|
161
|
+
except OSError:
|
|
162
|
+
log.warning("could not unlink stale socket %s", stale_sock, exc_info=True)
|
|
163
|
+
try:
|
|
164
|
+
self.server.start()
|
|
165
|
+
except Exception as exc: # noqa: BLE001 — socket bind failure is fatal-but-reported
|
|
166
|
+
print(f"jrag watch: failed to start server: {exc}", file=sys.stderr)
|
|
167
|
+
self.lock.release()
|
|
168
|
+
self._cleanup_runtime_files()
|
|
169
|
+
return 2
|
|
170
|
+
# 4b. Watcher start.
|
|
171
|
+
try:
|
|
172
|
+
self.watcher.start()
|
|
173
|
+
except Exception as exc: # noqa: BLE001 — reported; server already up so shut it down
|
|
174
|
+
print(f"jrag watch: failed to start watcher: {exc}", file=sys.stderr)
|
|
175
|
+
self.server.shutdown()
|
|
176
|
+
self._cleanup_runtime_files()
|
|
177
|
+
self.lock.release()
|
|
178
|
+
return 2
|
|
179
|
+
|
|
180
|
+
# 5. Write the initial state file (force, so --status sees truth at once).
|
|
181
|
+
self._state["started_at"] = time.time()
|
|
182
|
+
self._state["pid"] = os.getpid()
|
|
183
|
+
self._write_state()
|
|
184
|
+
|
|
185
|
+
# 6 + 7. Serve, then tear down. _shutdown ends with os._exit(0) so the
|
|
186
|
+
# finally never falls through; the return is unreachable.
|
|
187
|
+
try:
|
|
188
|
+
self._serve_until_stopped()
|
|
189
|
+
finally:
|
|
190
|
+
self._shutdown()
|
|
191
|
+
return 0 # pragma: no cover — os._exit in _shutdown
|
|
192
|
+
|
|
193
|
+
# ------------------------------------------------------------------
|
|
194
|
+
# event recording (called from the watcher debounce thread + the UI loop)
|
|
195
|
+
# ------------------------------------------------------------------
|
|
196
|
+
|
|
197
|
+
def _record(self, kind: str, detail: dict[str, Any]) -> None:
|
|
198
|
+
"""Update in-memory state from a watcher event; throttle state rewrites.
|
|
199
|
+
|
|
200
|
+
Called on the watcher's debounce worker thread, so all state mutation is
|
|
201
|
+
under ``_state_lock``. The state file is rewritten at most once per
|
|
202
|
+
``_STATE_WRITE_MIN_INTERVAL_S`` so ``--status`` readers see recent truth
|
|
203
|
+
without disk churn during a reindex burst.
|
|
204
|
+
"""
|
|
205
|
+
with self._state_lock:
|
|
206
|
+
if kind == "indexing_done":
|
|
207
|
+
self._state["last_reindex_at"] = time.time()
|
|
208
|
+
self._state["last_reindex_kind"] = "+".join(detail.get("kinds", []))
|
|
209
|
+
self._state["reindex_count"] += 1
|
|
210
|
+
elif kind == "indexing_started":
|
|
211
|
+
self._state["last_reindex_kind"] = (
|
|
212
|
+
"indexing:" + "+".join(detail.get("kinds", []))
|
|
213
|
+
)
|
|
214
|
+
elif kind == "error":
|
|
215
|
+
self._state["last_error"] = {
|
|
216
|
+
"phase": detail.get("phase"),
|
|
217
|
+
"at": time.time(),
|
|
218
|
+
"detail": detail,
|
|
219
|
+
}
|
|
220
|
+
self._maybe_write_state_locked()
|
|
221
|
+
|
|
222
|
+
# ------------------------------------------------------------------
|
|
223
|
+
# serve loop + status panel
|
|
224
|
+
# ------------------------------------------------------------------
|
|
225
|
+
|
|
226
|
+
def _serve_until_stopped(self) -> None:
|
|
227
|
+
"""Render the status panel and block until the stop flag is set.
|
|
228
|
+
|
|
229
|
+
On a non-TTY stdio (detached, piped, tests) the Live region is skipped in
|
|
230
|
+
favor of a single startup line — ``rich.Live`` on a pipe reprints the
|
|
231
|
+
whole panel on every update and would flood the redirect log.
|
|
232
|
+
"""
|
|
233
|
+
from rich.console import Console
|
|
234
|
+
|
|
235
|
+
console = Console()
|
|
236
|
+
live = None
|
|
237
|
+
if console.is_terminal:
|
|
238
|
+
try:
|
|
239
|
+
from rich.live import Live
|
|
240
|
+
|
|
241
|
+
live = Live(
|
|
242
|
+
self._render_panel(),
|
|
243
|
+
console=console,
|
|
244
|
+
refresh_per_second=4,
|
|
245
|
+
transient=False,
|
|
246
|
+
)
|
|
247
|
+
live.start()
|
|
248
|
+
except Exception: # noqa: BLE001 — Live is cosmetic; never block serving
|
|
249
|
+
live = None
|
|
250
|
+
if live is None:
|
|
251
|
+
print(
|
|
252
|
+
f"jrag watch: serving on {self._state['socket']} "
|
|
253
|
+
f"(pid {os.getpid()})",
|
|
254
|
+
flush=True,
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
try:
|
|
258
|
+
while not self._stop.is_set():
|
|
259
|
+
if live is not None:
|
|
260
|
+
try:
|
|
261
|
+
live.update(self._render_panel())
|
|
262
|
+
except Exception: # noqa: BLE001 — cosmetic
|
|
263
|
+
pass
|
|
264
|
+
# Event.wait returns True as soon as the flag is set, so a stop
|
|
265
|
+
# signal is observed within one tick rather than the full window.
|
|
266
|
+
self._stop.wait(_LOOP_TICK_S)
|
|
267
|
+
finally:
|
|
268
|
+
if live is not None:
|
|
269
|
+
try:
|
|
270
|
+
live.stop()
|
|
271
|
+
except Exception: # noqa: BLE001 — cosmetic
|
|
272
|
+
pass
|
|
273
|
+
|
|
274
|
+
def _render_panel(self):
|
|
275
|
+
"""Build the ``rich`` status table from the current in-memory state."""
|
|
276
|
+
from rich.table import Table
|
|
277
|
+
|
|
278
|
+
with self._state_lock:
|
|
279
|
+
state = dict(self._state)
|
|
280
|
+
table = Table(title=f"jrag watch (pid {os.getpid()})", show_header=False, box=None)
|
|
281
|
+
table.add_row("socket", str(state.get("socket")))
|
|
282
|
+
table.add_row("reindex count", str(state.get("reindex_count", 0)))
|
|
283
|
+
last_kind = state.get("last_reindex_kind")
|
|
284
|
+
last_at = state.get("last_reindex_at")
|
|
285
|
+
if last_kind:
|
|
286
|
+
when = time.strftime("%H:%M:%S", time.localtime(last_at)) if last_at else "—"
|
|
287
|
+
table.add_row("last reindex", f"{last_kind} ({when})")
|
|
288
|
+
else:
|
|
289
|
+
table.add_row("last reindex", "—")
|
|
290
|
+
return table
|
|
291
|
+
|
|
292
|
+
# ------------------------------------------------------------------
|
|
293
|
+
# shutdown
|
|
294
|
+
# ------------------------------------------------------------------
|
|
295
|
+
|
|
296
|
+
def _shutdown(self) -> None:
|
|
297
|
+
"""Tear down watcher → server → lock, remove runtime files, ``os._exit(0)``.
|
|
298
|
+
|
|
299
|
+
Each step is best-effort: a failure in one must not skip the rest, and
|
|
300
|
+
the process MUST terminate via ``os._exit(0)`` (never a normal return)
|
|
301
|
+
to avoid the lance worker-thread SIGABRT at finalization once the server
|
|
302
|
+
has served a ``search`` query.
|
|
303
|
+
"""
|
|
304
|
+
try:
|
|
305
|
+
self.watcher.stop()
|
|
306
|
+
except Exception: # noqa: BLE001 — teardown must continue
|
|
307
|
+
log.warning("watcher.stop raised during shutdown", exc_info=True)
|
|
308
|
+
try:
|
|
309
|
+
self.server.shutdown()
|
|
310
|
+
except Exception: # noqa: BLE001
|
|
311
|
+
log.warning("server.shutdown raised during shutdown", exc_info=True)
|
|
312
|
+
try:
|
|
313
|
+
self.lock.release()
|
|
314
|
+
except Exception: # noqa: BLE001
|
|
315
|
+
log.warning("lock.release raised during shutdown", exc_info=True)
|
|
316
|
+
self._cleanup_runtime_files()
|
|
317
|
+
sys.stdout.flush()
|
|
318
|
+
sys.stderr.flush()
|
|
319
|
+
os._exit(0)
|
|
320
|
+
|
|
321
|
+
# ------------------------------------------------------------------
|
|
322
|
+
# helpers
|
|
323
|
+
# ------------------------------------------------------------------
|
|
324
|
+
|
|
325
|
+
def _on_signal(self, signum, frame) -> None: # noqa: ARG002 — signal API
|
|
326
|
+
"""SIGINT/SIGTERM handler: flag the serve loop to stop (main thread)."""
|
|
327
|
+
self._stop.set()
|
|
328
|
+
|
|
329
|
+
def _cleanup_runtime_files(self) -> None:
|
|
330
|
+
"""Remove the socket and state file (idempotent, best-effort)."""
|
|
331
|
+
for path in (
|
|
332
|
+
paths.socket_path(self.cfg.index_dir),
|
|
333
|
+
paths.state_path(self.cfg.index_dir),
|
|
334
|
+
):
|
|
335
|
+
try:
|
|
336
|
+
path.unlink()
|
|
337
|
+
except FileNotFoundError:
|
|
338
|
+
pass
|
|
339
|
+
except OSError:
|
|
340
|
+
log.warning("could not unlink %s", path, exc_info=True)
|
|
341
|
+
|
|
342
|
+
def _write_state(self) -> None:
|
|
343
|
+
"""Unconditionally write the state JSON now (initial write).
|
|
344
|
+
|
|
345
|
+
Acquires ``_state_lock``; the throttled re-write path is
|
|
346
|
+
:meth:`_maybe_write_state_locked`, called by :meth:`_record` which
|
|
347
|
+
already holds the lock.
|
|
348
|
+
"""
|
|
349
|
+
with self._state_lock:
|
|
350
|
+
self._write_state_locked()
|
|
351
|
+
|
|
352
|
+
def _maybe_write_state_locked(self) -> None:
|
|
353
|
+
"""Throttled state write; caller MUST hold ``_state_lock``."""
|
|
354
|
+
now = time.monotonic()
|
|
355
|
+
if now - self._last_state_write >= _STATE_WRITE_MIN_INTERVAL_S:
|
|
356
|
+
self._write_state_locked()
|
|
357
|
+
|
|
358
|
+
def _write_state_locked(self) -> None:
|
|
359
|
+
"""Write the JSON state file (best-effort); caller MUST hold ``_state_lock``."""
|
|
360
|
+
path = paths.state_path(self.cfg.index_dir)
|
|
361
|
+
data = dict(self._state)
|
|
362
|
+
try:
|
|
363
|
+
tmp = path.with_suffix(path.suffix + ".tmp")
|
|
364
|
+
tmp.write_text(json.dumps(data))
|
|
365
|
+
os.replace(tmp, path)
|
|
366
|
+
self._last_state_write = time.monotonic()
|
|
367
|
+
except OSError:
|
|
368
|
+
log.warning("could not write state file %s", path, exc_info=True)
|