dirsql 0.3.65 → 0.3.67

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,268 +0,0 @@
1
- ---
2
- canonical: https://thekevinscott.github.io/dirsql/guide/async
3
- ---
4
-
5
- # Async API
6
-
7
- > Online: <https://thekevinscott.github.io/dirsql/guide/async>
8
-
9
- `DirSQL` is async by default in Python. The initial directory scan runs in a background thread so it does not block the event loop.
10
-
11
- ## Basic usage
12
-
13
- ::: code-group
14
-
15
- ```python [Python]
16
- import asyncio
17
- import json
18
- from dirsql import DirSQL, Table
19
-
20
- async def main():
21
- db = DirSQL(
22
- "./my-project",
23
- tables=[
24
- Table(
25
- ddl="CREATE TABLE items (name TEXT, value INTEGER)",
26
- glob="data/*.json",
27
- extract=lambda path: [json.loads(open(path, encoding="utf-8").read())],
28
- ),
29
- ],
30
- )
31
- await db.ready()
32
-
33
- # Query (runs in a thread, does not block the event loop)
34
- results = await db.query("SELECT * FROM items WHERE value > 10")
35
- print(results)
36
-
37
- asyncio.run(main())
38
- ```
39
-
40
- ```rust [Rust]
41
- use dirsql::{DirSQL, Table, Value};
42
- use std::collections::HashMap;
43
-
44
- // See `row_from_json` in getting-started.md for a reusable helper that
45
- // turns a JSON object into a dirsql row (dirsql::Value is not Deserialize,
46
- // so a row can't be produced by serde_json::from_str directly).
47
- fn row_from_json(raw: &str) -> HashMap<String, Value> {
48
- let v: serde_json::Value = serde_json::from_str(raw).unwrap();
49
- let serde_json::Value::Object(obj) = v else { return HashMap::new() };
50
- obj.into_iter()
51
- .map(|(k, val)| {
52
- let v = match val {
53
- serde_json::Value::String(s) => Value::Text(s),
54
- serde_json::Value::Number(n) => n
55
- .as_i64()
56
- .map(Value::Integer)
57
- .unwrap_or_else(|| Value::Real(n.as_f64().unwrap_or(0.0))),
58
- serde_json::Value::Bool(b) => Value::Integer(b as i64),
59
- serde_json::Value::Null => Value::Null,
60
- other => Value::Text(other.to_string()),
61
- };
62
- (k, v)
63
- })
64
- .collect()
65
- }
66
-
67
- #[tokio::main]
68
- async fn main() -> Result<(), Box<dyn std::error::Error>> {
69
- let db = DirSQL::new(
70
- "./my-project",
71
- vec![
72
- Table::new(
73
- "CREATE TABLE items (name TEXT, value INTEGER)",
74
- "data/*.json",
75
- |path| vec![row_from_json(&std::fs::read_to_string(path).unwrap())],
76
- ),
77
- ],
78
- )?;
79
-
80
- let results = db.query("SELECT * FROM items WHERE value > 10")?;
81
- println!("{:?}", results);
82
- Ok(())
83
- }
84
- ```
85
-
86
- ```typescript [TypeScript]
87
- import { readFileSync } from 'node:fs';
88
- import { DirSQL, Table } from 'dirsql';
89
-
90
- const db = new DirSQL({
91
- root: './my-project',
92
- tables: [
93
- new Table({
94
- ddl: 'CREATE TABLE items (name TEXT, value INTEGER)',
95
- glob: 'data/*.json',
96
- extract: (path) => [JSON.parse(readFileSync(path, 'utf8'))],
97
- }),
98
- ],
99
- });
100
-
101
- // Query
102
- const results = await db.query('SELECT * FROM items WHERE value > 10');
103
- console.log(results);
104
- ```
105
-
106
- :::
107
-
108
- ## Constructor
109
-
110
- ```python
111
- DirSQL(root=None, *, tables=None, ignore=None, config=None, persist=False, persist_path=None)
112
- ```
113
-
114
- The constructor immediately starts scanning in a background thread via `asyncio.ensure_future`. The constructor itself returns immediately without blocking.
115
-
116
- ## `await db.ready()`
117
-
118
- Waits until the initial directory scan is complete. If the scan raised an exception (invalid DDL, unreadable files, etc.), `ready()` re-raises it.
119
-
120
- `ready()` can be called multiple times safely. After the first completion, subsequent calls return immediately.
121
-
122
- ```python
123
- db = DirSQL("./data", tables=[...])
124
-
125
- # Do other setup work here while the scan runs in the background
126
- setup_logging()
127
- connect_websocket()
128
-
129
- # Now wait for the scan to finish before querying
130
- await db.ready()
131
- ```
132
-
133
- ## `await db.query(sql)`
134
-
135
- Runs a SQL query in a background thread. Returns a list of dicts keyed by column name.
136
-
137
- ```python
138
- results = await db.query("SELECT COUNT(*) as n FROM items")
139
- ```
140
-
141
- ## `async for event in db.watch()`
142
-
143
- Returns an async iterable of `RowEvent` objects. The watcher is started automatically on the first iteration.
144
-
145
- ::: code-group
146
-
147
- ```python [Python]
148
- async for event in db.watch():
149
- if event.action == "insert":
150
- print(f"New row in {event.table}: {event.row}")
151
- elif event.action == "update":
152
- print(f"Updated row in {event.table}: {event.row}")
153
- elif event.action == "delete":
154
- print(f"Deleted row from {event.table}: {event.row}")
155
- elif event.action == "error":
156
- print(f"Error: {event.error}")
157
- ```
158
-
159
- ```rust [Rust]
160
- // `RowEvent` is an enum; match on the variant to destructure its fields.
161
- // `StreamExt` (for `.next()`) comes from the `futures` crate, which is only a
162
- // dirsql dependency under its `cli` feature -- add it to your own project:
163
- //
164
- // cargo add futures
165
- use dirsql::RowEvent;
166
- use futures::StreamExt;
167
-
168
- let mut stream = db.watch()?; // watch() returns Result<WatchStream>
169
- while let Some(event) = stream.next().await {
170
- match event {
171
- RowEvent::Insert { table, row, file_path } => {
172
- println!("New row in {table} ({file_path}): {row:?}")
173
- }
174
- RowEvent::Update { table, new_row, file_path, .. } => {
175
- println!("Updated row in {table} ({file_path}): {new_row:?}")
176
- }
177
- RowEvent::Delete { table, row, file_path } => {
178
- println!("Deleted row from {table} ({file_path}): {row:?}")
179
- }
180
- RowEvent::Error { file_path, error, .. } => {
181
- eprintln!("Error on {file_path:?}: {error}")
182
- }
183
- }
184
- }
185
- ```
186
-
187
- ```typescript [TypeScript]
188
- for await (const event of db.watch()) {
189
- switch (event.action) {
190
- case 'insert':
191
- console.log(`New row in ${event.table}:`, event.row);
192
- break;
193
- case 'update':
194
- console.log(`Updated row in ${event.table}:`, event.row);
195
- break;
196
- case 'delete':
197
- console.log(`Deleted row from ${event.table}:`, event.row);
198
- break;
199
- case 'error':
200
- console.error(`Error: ${event.error}`);
201
- break;
202
- }
203
- }
204
- ```
205
-
206
- :::
207
-
208
- The async iterator polls for filesystem events with a 200ms timeout internally. It yields events as they arrive and never terminates on its own -- use `break` or cancellation to stop.
209
-
210
- ## Combining with other async code
211
-
212
- The async API works alongside other concurrent code without blocking:
213
-
214
- ::: code-group
215
-
216
- ```python [Python]
217
- async def watch_and_serve(db):
218
- async for event in db.watch():
219
- await notify_clients(event)
220
-
221
- async def main():
222
- db = DirSQL("./data", tables=[...])
223
- await asyncio.gather(
224
- watch_and_serve(db),
225
- run_web_server(),
226
- )
227
- ```
228
-
229
- ```rust [Rust]
230
- // `.next()` needs `StreamExt` from the `futures` crate (`cargo add futures`).
231
- use futures::StreamExt;
232
-
233
- async fn watch_and_serve(db: &DirSQL) -> Result<(), Box<dyn std::error::Error>> {
234
- let mut stream = db.watch()?; // watch() returns Result<WatchStream>
235
- while let Some(event) = stream.next().await {
236
- notify_clients(&event).await;
237
- }
238
- Ok(())
239
- }
240
-
241
- #[tokio::main]
242
- async fn main() -> Result<(), Box<dyn std::error::Error>> {
243
- let db = DirSQL::new("./data", vec![/* tables */])?;
244
-
245
- tokio::join!(
246
- watch_and_serve(&db),
247
- run_web_server(),
248
- );
249
- Ok(())
250
- }
251
- ```
252
-
253
- ```typescript [TypeScript]
254
- async function watchAndServe(db: DirSQL) {
255
- for await (const event of db.watch()) {
256
- await notifyClients(event);
257
- }
258
- }
259
-
260
- const db = new DirSQL({ root: './data', tables: [/* tables */] });
261
-
262
- await Promise.all([
263
- watchAndServe(db),
264
- runWebServer(),
265
- ]);
266
- ```
267
-
268
- :::
@@ -1,161 +0,0 @@
1
- ---
2
- canonical: https://thekevinscott.github.io/dirsql/guide/crdt
3
- ---
4
-
5
- # Collaboration with CRDTs
6
-
7
- > Online: <https://thekevinscott.github.io/dirsql/guide/crdt>
8
-
9
- `dirsql` treats the filesystem as the source of truth. That works well when a single process (or a single human) is writing, but breaks down for multi-writer collaboration: two peers editing the same file concurrently produce a merge conflict, not a merged result.
10
-
11
- [Conflict-free Replicated Data Types](https://crdt.tech/) (CRDTs) solve that merge problem at the data-structure level, not the filesystem level. Two replicas that apply the same set of edits -- in any order, with any network partitions in between -- converge on the same final state, without a central arbiter.
12
-
13
- This guide is **opinionated**. It picks one library, explains the integration pattern with `dirsql`, and names the alternatives so you can steer if your situation is different.
14
-
15
- ## Recommendation: Automerge
16
-
17
- Use [Automerge](https://automerge.org/) (specifically the 2.x series with [automerge-repo](https://automerge.org/automerge-repo/) for sync).
18
-
19
- Why Automerge over the alternatives:
20
-
21
- - **JSON-shaped document model.** Automerge docs look like nested maps, lists, and text. That matches `dirsql`'s one-object-per-file workflow -- each Automerge document is the thing your `extract` function projects into rows.
22
- - **Cross-language SDKs that mirror `dirsql`'s parity story.** First-class Rust ([`automerge`](https://crates.io/crates/automerge)), TypeScript ([`@automerge/automerge`](https://www.npmjs.com/package/@automerge/automerge)), and Python ([`automerge`](https://pypi.org/project/automerge/)) implementations exist today, all driven by the same Rust core. If you already have all three `dirsql` SDKs in play, Automerge won't force a language monoculture on you.
23
- - **Filesystem-friendly sync primitives.** `automerge-repo` ships a [`NodeFSStorageAdapter`](https://automerge.org/docs/repositories/storage/) that shards document history into regular files under a directory. That directory is exactly the kind of tree `dirsql` is designed to watch.
24
- - **Binary format with a deterministic JSON view.** You never hand-edit a CRDT file, but every replica projects the same canonical JSON from the binary state. That canonical JSON is what you feed to `dirsql`'s extract function, so two peers that have synced will produce identical rows.
25
-
26
- ## The integration shape
27
-
28
- There are two files per logical "document":
29
-
30
- ```
31
- workspace/
32
- posts/
33
- hello/
34
- doc.automerge <-- binary CRDT state (the source of truth for writers)
35
- view.json <-- materialized JSON snapshot (written on each merge)
36
- ```
37
-
38
- - Writers (editors, sync peers, etc.) mutate `doc.automerge` through Automerge APIs.
39
- - After every mutation, the writer serializes the current document to `view.json`. This is the file `dirsql` indexes.
40
- - `dirsql` watches `view.json`, not `doc.automerge`. The CRDT file is an implementation detail of how the JSON got there.
41
-
42
- This keeps `dirsql`'s extract function oblivious to CRDT semantics: it reads a plain JSON file, exactly as it would without Automerge.
43
-
44
- ::: tip Why not `extract` directly from `.automerge`?
45
- You could -- the Rust and Python Automerge SDKs let you load a binary doc and walk its fields. But it couples your table schema to the CRDT library version, makes `extract` non-pure (it allocates CRDT state on every file change), and buys nothing: the writer is the only place that can produce a valid Automerge blob, so it might as well produce the JSON view at the same time.
46
- :::
47
-
48
- ### Example: posts as Automerge documents
49
-
50
- ::: code-group
51
-
52
- ```python [Python]
53
- from dirsql import DirSQL, Table
54
- import json
55
-
56
- db = DirSQL(
57
- "./workspace",
58
- tables=[
59
- Table(
60
- ddl="CREATE TABLE posts (id TEXT, title TEXT, body TEXT, updated INTEGER)",
61
- # Match the JSON view, not the raw CRDT binary.
62
- glob="posts/*/view.json",
63
- extract=lambda path: [json.loads(open(path, encoding="utf-8").read())],
64
- ),
65
- ],
66
- )
67
- ```
68
-
69
- ```rust [Rust]
70
- use dirsql::{DirSQL, Table};
71
- // See `row_from_json` in getting-started.md.
72
-
73
- let db = DirSQL::new(
74
- "./workspace",
75
- vec![
76
- Table::new(
77
- "CREATE TABLE posts (id TEXT, title TEXT, body TEXT, updated INTEGER)",
78
- "posts/*/view.json",
79
- |path| vec![row_from_json(&std::fs::read_to_string(path).unwrap())],
80
- ),
81
- ],
82
- )?;
83
- ```
84
-
85
- ```typescript [TypeScript]
86
- import { readFileSync } from 'node:fs';
87
- import { DirSQL, type TableDef } from 'dirsql';
88
-
89
- const tables: TableDef[] = [
90
- {
91
- ddl: 'CREATE TABLE posts (id TEXT, title TEXT, body TEXT, updated INTEGER)',
92
- glob: 'posts/*/view.json',
93
- extract: (path) => [JSON.parse(readFileSync(path, 'utf8'))],
94
- },
95
- ];
96
-
97
- const db = new DirSQL({ root: './workspace', tables });
98
- ```
99
-
100
- :::
101
-
102
- The Automerge writer (sketch, TypeScript):
103
-
104
- ```typescript
105
- import * as Automerge from '@automerge/automerge';
106
- import { writeFileSync, readFileSync } from 'node:fs';
107
-
108
- // Load (or create) the CRDT doc.
109
- const bytes = readFileSync('workspace/posts/hello/doc.automerge');
110
- let doc = Automerge.load<Post>(bytes);
111
-
112
- // Apply an edit.
113
- doc = Automerge.change(doc, (d) => {
114
- d.title = 'Hello, world';
115
- d.updated = Date.now();
116
- });
117
-
118
- // Persist both the CRDT state and the materialized view.
119
- writeFileSync('workspace/posts/hello/doc.automerge', Automerge.save(doc));
120
- writeFileSync('workspace/posts/hello/view.json', JSON.stringify(doc));
121
- ```
122
-
123
- `dirsql`'s watcher picks up the change to `view.json`, re-runs `extract`, and emits an `update` row event. Queries over `posts` reflect the merged state without `dirsql` knowing Automerge exists.
124
-
125
- ## Multi-writer, in practice
126
-
127
- 1. Each peer runs an `automerge-repo` instance with the filesystem storage adapter pointed at its local `workspace/`.
128
- 2. Peers sync via any transport `automerge-repo` supports ([WebSocket](https://automerge.org/docs/repositories/networking/), [BroadcastChannel](https://automerge.org/docs/repositories/networking/), or a custom adapter).
129
- 3. On every sync, the repo applies incoming ops to the local CRDT, writes the updated `doc.automerge`, and the writer code re-serializes `view.json`.
130
- 4. Every peer's `dirsql` sees the same eventual `view.json` and produces the same rows.
131
-
132
- The key invariant: **`view.json` is a deterministic projection of `doc.automerge`**. Two peers that have converged on the CRDT state must produce byte-identical (or at least semantically-identical) JSON views. Otherwise you get spurious `update` events that flap with sync order. Use `JSON.stringify` with sorted keys (or `json.dumps(..., sort_keys=True)` in Python) to guarantee this.
133
-
134
- ## Tradeoffs vs plain files
135
-
136
- When **plain files** are the right answer:
137
-
138
- - Single writer. A solo user editing `posts/*.json` will never hit a merge conflict. Adding a CRDT is overhead.
139
- - Human-readable history matters. `git diff` on a JSON file tells a story; `git diff` on a CRDT binary does not.
140
- - Schema churn is frequent. Renaming a field in plain JSON is a `sed`; in a CRDT it's a migration.
141
-
142
- When **CRDTs** earn their complexity:
143
-
144
- - Multi-writer without a central server (local-first, peer-to-peer).
145
- - Offline edits that need to merge on reconnect.
146
- - Fine-grained collaborative editing (cursor-level merging of a shared text field).
147
-
148
- Hybrid is common: keep configuration and reference data as plain files, and use CRDTs only for the documents that genuinely have multiple writers.
149
-
150
- ## Alternatives we considered
151
-
152
- - [**Yjs**](https://docs.yjs.dev/) -- the dominant JS CRDT, excellent for rich-text collaboration (it backs many of the production collab editors you've used). Skipped as the primary recommendation because its Rust port ([`yrs`](https://crates.io/crates/yrs)) and Python bindings lag the JS implementation. If your workload is browser-first and text-heavy, prefer Yjs.
153
- - [**Loro**](https://loro.dev/) -- Rust-first CRDT with a clean API and good cross-language story. Worth watching; we'd consider it once its Python bindings are GA. Try it if you're Rust-centric and don't need Automerge's ecosystem.
154
- - **Operational Transform / hand-rolled merge logic** -- don't. OT is correct but hard to implement right, and you lose the offline-peer story that CRDTs give you for free.
155
- - **Git as the merge engine** -- tempting because `dirsql` already lives on the filesystem, but three-way merges of structured JSON produce garbage conflict markers that no extract function can parse. Use a CRDT.
156
-
157
- ## See also
158
-
159
- - [Ink & Switch's local-first essay](https://www.inkandswitch.com/local-first/) -- the design space CRDTs sit in.
160
- - [Automerge documentation](https://automerge.org/docs/) -- API reference and sync-adapter guides.
161
- - [`crdt.tech`](https://crdt.tech/) -- library survey across languages.
@@ -1,177 +0,0 @@
1
- # Persistence
2
-
3
- By default `dirsql` keeps its SQLite database in memory and rebuilds it from scratch every time the process starts. For large directories this can take seconds to minutes -- nearly all of which is spent re-parsing files that haven't changed since the previous run.
4
-
5
- Persistence stores the SQLite database on disk so that subsequent startups only re-parse the files that have actually changed.
6
-
7
- ::: tip Same answers, faster startup
8
- The rows returned by `query()` after a persistent startup are equivalent to those produced by a from-scratch rebuild. Persistence is a startup-time optimization, not a correctness compromise. The reconcile algorithm is the same one `git status` uses to decide which files have changed since the last index write.
9
- :::
10
-
11
- ## Quick start
12
-
13
- ::: code-group
14
-
15
- ```toml [.dirsql.toml]
16
- [dirsql]
17
- persist = true
18
- ```
19
-
20
- ```python [Python]
21
- from dirsql import DirSQL
22
-
23
- db = DirSQL("./my-project", tables=[...], persist=True)
24
- await db.ready()
25
- ```
26
-
27
- ```rust [Rust]
28
- use dirsql::DirSQL;
29
-
30
- let db = DirSQL::builder()
31
- .root("./my-project")
32
- .tables(vec![/* ... */])
33
- .persist(true)
34
- .build()?;
35
- ```
36
-
37
- ```typescript [TypeScript]
38
- import { DirSQL } from "dirsql";
39
-
40
- const db = new DirSQL({ root: "./my-project", tables: [/* ... */], persist: true });
41
- await db.ready;
42
- ```
43
-
44
- :::
45
-
46
- That's it. The first run writes the database to `./my-project/.dirsql/cache.db`. Every subsequent startup uses the cache.
47
-
48
- ## Configuration
49
-
50
- | Option | Type | Default | Meaning |
51
- |---|---|---|---|
52
- | `persist` | boolean | `false` | Enable persistent on-disk storage. |
53
- | `persist_path` (Python, Rust) / `persistPath` (TypeScript) | string | `<root>/.dirsql/cache.db` | Override the database file path. Ignored when `persist` is `false`. |
54
-
55
- The default location keeps the cache alongside the data it indexes, which means it follows the project around (clone, copy, move) without extra setup. Override `persist_path` if you want the cache somewhere else -- a CI cache directory, a tmpfs mount, an XDG cache dir, etc.
56
-
57
- ::: code-group
58
-
59
- ```toml [.dirsql.toml]
60
- [dirsql]
61
- persist = true
62
- persist_path = "/var/cache/dirsql/myproject.db"
63
- ```
64
-
65
- ```python [Python]
66
- db = DirSQL(
67
- "./my-project",
68
- tables=[...],
69
- persist=True,
70
- persist_path="/var/cache/dirsql/myproject.db",
71
- )
72
- ```
73
-
74
- ```rust [Rust]
75
- let db = DirSQL::builder()
76
- .root("./my-project")
77
- .tables(vec![/* ... */])
78
- .persist(true)
79
- .persist_path("/var/cache/dirsql/myproject.db")
80
- .build()?;
81
- ```
82
-
83
- ```typescript [TypeScript]
84
- const db = new DirSQL({
85
- root: "./my-project",
86
- tables: [/* ... */],
87
- persist: true,
88
- persistPath: "/var/cache/dirsql/myproject.db",
89
- });
90
- ```
91
-
92
- :::
93
-
94
- ## The `.dirsql/` directory
95
-
96
- `dirsql` reserves the top-level `.dirsql/` directory inside every scanned root. It is **unconditionally excluded from the directory walk**, whether persistence is enabled or not. This means:
97
-
98
- - The default cache path `<root>/.dirsql/cache.db` cannot accidentally be ingested as a data file.
99
- - You can place additional `dirsql`-related files in `.dirsql/` (e.g. a project-local config snapshot) without them being parsed.
100
- - You should not put your own data files in `.dirsql/` -- they will be silently ignored.
101
-
102
- If you persist into `.dirsql/`, add it to your `.gitignore`:
103
-
104
- ```
105
- .dirsql/
106
- ```
107
-
108
- The cache file should never be committed -- it is reproducible from the source tree and frequently large.
109
-
110
- ## How the startup reconcile works
111
-
112
- When a persistent cache exists, `dirsql` does not blindly trust it. On startup it:
113
-
114
- 1. **Checks compatibility metadata.** If the cached `dirsql` version, schema version, glob configuration, parser versions, or canonical root path differs from the current build, the cache is wiped and rebuilt from scratch.
115
- 2. **Walks the tree and stats every matching file.** This is metadata-only -- no file contents are read.
116
- 3. **For each file, compares the live `(size, mtime, ctime, inode, dev)` tuple against the cached row:**
117
- - **Trust the cache** when every field matches *and* the file's mtime is older than the cache's snapshot time (outside the racy window).
118
- - **Hash-confirm** when the tuple matches but the file's mtime falls inside the racy window. `dirsql` reads and hashes the file; if the hash matches the cached hash, the cache is trusted.
119
- - **Re-parse** when any field of the tuple differs.
120
- 4. **Deletes** rows for files that were in the cache but are no longer on disk.
121
- 5. **Inserts** rows for files that are on disk but were not in the cache.
122
-
123
- This is the same algorithm `git status` uses to decide which files have changed since the last index write. The "racy window" handling is what closes the gap when a file is modified within the same filesystem-timestamp resolution as the cache write.
124
-
125
- ## When `dirsql` does a full rebuild
126
-
127
- Any of the following will cause the cache to be discarded and rebuilt from scratch on the next startup:
128
-
129
- - The `dirsql` library was upgraded between runs.
130
- - The glob configuration changed (a new table, a removed table, a modified glob, a changed `ignore` list).
131
- - A built-in parser version changed (this generally only happens on `dirsql` upgrades).
132
- - The cache was written for a different root directory than the one currently configured.
133
- - The internal schema of the cache changed (i.e. you upgraded `dirsql` across a schema version bump).
134
-
135
- Full rebuilds take exactly as long as a non-persistent startup -- there is no penalty for them, only a missed optimization.
136
-
137
- ## Limitations
138
-
139
- ### Network filesystems
140
-
141
- NFS, SMB/CIFS, and similar network filesystems cache file attributes on the client and can return stale `stat` results. Persistent mode is **not supported** on network filesystems and may produce stale rows. Use in-memory mode (the default) if your `root` lives on a network mount.
142
-
143
- ### The mtime-preservation edge case
144
-
145
- Racy-stat detection misses changes only when **all** of the following are true:
146
-
147
- - A file's contents are modified.
148
- - The file's size after modification is identical to its size before.
149
- - The file's `mtime` is externally reset to a value older than the cache's snapshot time (e.g. via `touch -r` or a backup-restore tool that preserves mtime).
150
-
151
- If you cannot tolerate this edge case, disable persistence (`persist = false`). This is the same trade-off `git` makes with `core.trustctime` / `core.checkStat`.
152
-
153
- ### Single writer
154
-
155
- Only one `dirsql` process should write to a given cache file at a time. Multiple read-only processes can query the same file safely once the writer finishes the initial reconcile. Coordinated multi-writer access is not supported in v0.3.0.
156
-
157
- ## Inspecting the cache
158
-
159
- The persistent database is a normal SQLite file. You can open it with any SQLite client:
160
-
161
- ```bash
162
- sqlite3 .dirsql/cache.db
163
- ```
164
-
165
- ```sql
166
- .tables
167
- -- comments documents metrics _dirsql_files _dirsql_meta
168
-
169
- SELECT * FROM _dirsql_meta;
170
- -- schema_version | 1
171
- -- dirsql_version | 0.3.0
172
- -- glob_config_hash | <hex>
173
- -- parser_versions | {"json":"1","jsonl":"1","csv":"1",...}
174
- -- root_canonical | /home/alice/my-project
175
- ```
176
-
177
- The `_dirsql_files` and `_dirsql_meta` tables are managed by `dirsql`. Do not modify them by hand -- on the next startup, `dirsql` will detect the inconsistency and rebuild from scratch.