holdings 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holdings-0.1.0/LICENSE +28 -0
- holdings-0.1.0/PKG-INFO +118 -0
- holdings-0.1.0/README.md +102 -0
- holdings-0.1.0/holdings.egg-info/PKG-INFO +118 -0
- holdings-0.1.0/holdings.egg-info/SOURCES.txt +10 -0
- holdings-0.1.0/holdings.egg-info/dependency_links.txt +1 -0
- holdings-0.1.0/holdings.egg-info/entry_points.txt +2 -0
- holdings-0.1.0/holdings.egg-info/requires.txt +3 -0
- holdings-0.1.0/holdings.egg-info/top_level.txt +1 -0
- holdings-0.1.0/holdings.py +587 -0
- holdings-0.1.0/pyproject.toml +27 -0
- holdings-0.1.0/setup.cfg +4 -0
holdings-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Peter Foldiak
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
holdings-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: holdings
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Disposable, regenerable catalog of which media hold which files
|
|
5
|
+
Author: Peter Foldiak
|
|
6
|
+
License: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/petfold/holdings
|
|
8
|
+
Project-URL: Issues, https://github.com/petfold/holdings/issues
|
|
9
|
+
Keywords: backup,catalog,placement,redundancy,sqlite,content-addressed,ontodag
|
|
10
|
+
Requires-Python: >=3.9
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Provides-Extra: test
|
|
14
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# holdings — v0.1 of the placement catalog
|
|
18
|
+
|
|
19
|
+
A disposable, regenerable catalog answering: **which media hold which files?**
|
|
20
|
+
Single file, stdlib only, Python 3.9+. SQLite for placement truth; semantics
|
|
21
|
+
belong to OntoDAG (see *Projection contract* below).
|
|
22
|
+
|
|
23
|
+
## Design contract (the important part)
|
|
24
|
+
|
|
25
|
+
1. **Observer, not authority.** holdings only *reads* filesystems and backup
|
|
26
|
+
listings. It never writes to your data, never sits in the backup or sync
|
|
27
|
+
path. Deleting holdings and its database costs nothing but convenience.
|
|
28
|
+
2. **Everything is regenerable by re-scanning.** The catalog is a cache of
|
|
29
|
+
facts about the world. The only original data in the whole system is your
|
|
30
|
+
*human* OntoDAG categorization — which holdings never touches.
|
|
31
|
+
3. **Content hash is identity.** `sha256:…` is the primary key everywhere.
|
|
32
|
+
Paths, media, snapshots, categories: all attributes of a hash.
|
|
33
|
+
4. **Single writer, many readers.** Scan on the backup-node laptop. Put the
|
|
34
|
+
SQLite file in a Syncthing folder; every device then carries the full
|
|
35
|
+
index of everything — including drives offline in another country.
|
|
36
|
+
|
|
37
|
+
## Quick start
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
export HOLDINGS_DB=~/Sync/catalog/catalog.sqlite # put it in a synced folder
|
|
41
|
+
|
|
42
|
+
# Register your media once:
|
|
43
|
+
./holdings.py add-medium laptop-x1 --kind laptop --location "with me"
|
|
44
|
+
./holdings.py add-medium drive-budapest --kind drive --backup --location "safe, Budapest"
|
|
45
|
+
./holdings.py add-medium drive-standrews --kind drive --backup --location "office, St Andrews"
|
|
46
|
+
./holdings.py add-medium restic-b2 --kind restic-repo --backup
|
|
47
|
+
|
|
48
|
+
# Scan whenever a medium is mounted (fast on rescan: unchanged files
|
|
49
|
+
# are recognized by size+mtime and not rehashed):
|
|
50
|
+
./holdings.py scan drive-budapest /media/peter/backup-drive
|
|
51
|
+
./holdings.py scan laptop-x1 /home/peter --exclude-file ~/backup/excludes.txt
|
|
52
|
+
|
|
53
|
+
# Count backup snapshots as copies (approximate matching by name+size;
|
|
54
|
+
# for exact hashes, `restic mount` the repo and `scan` it instead):
|
|
55
|
+
restic -r b2:bucket:repo ls --json latest | ./holdings.py import-restic restic-b2 -
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Queries
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
./holdings.py whereis holiday.jpg # every medium+path holding this content
|
|
62
|
+
./holdings.py whereis sha256:45887c... # by hash
|
|
63
|
+
./holdings.py redundancy --min-copies 2 # content below 2 backup copies
|
|
64
|
+
./holdings.py only-on drive-budapest # DANGER LIST: exists nowhere else
|
|
65
|
+
./holdings.py diff drive-a drive-b # on A but not B
|
|
66
|
+
./holdings.py media # media overview
|
|
67
|
+
./holdings.py stats # totals
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
`redundancy` turns your 3-2-1 policy into a checkable report.
|
|
71
|
+
`only-on` is the consolidation to-do list for old scattered drives: run it,
|
|
72
|
+
back those files up via restic, rescan, watch the list empty, then wipe the
|
|
73
|
+
drive with confidence.
|
|
74
|
+
|
|
75
|
+
## Projection contract (OntoDAG integration)
|
|
76
|
+
|
|
77
|
+
*(2026-08-20: this contract's canonical statement now lives at the meet
|
|
78
|
+
point — [ontodag `docs/plans/PROJECTIONS.md`](https://github.com/petfold/ontodag/blob/main/docs/plans/PROJECTIONS.md)
|
|
79
|
+
— which generalizes it across sources (files here, messages in ucomm)
|
|
80
|
+
and adds retention classes. The rules below remain the agreed file-side
|
|
81
|
+
instance and the wire format is unchanged.)*
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
./holdings.py project-ontodag --out placement.jsonl
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Emits JSON lines: `{"item": "<hash>", "supercategories": ["sys:on:<medium>",
|
|
88
|
+
"sys:type:<ext>", "sys:backup:<n>"]}` — matching OntoDAG's
|
|
89
|
+
`put(item, supercategories)` model.
|
|
90
|
+
|
|
91
|
+
The agreed rules for the ingesting side:
|
|
92
|
+
|
|
93
|
+
* everything under the **`sys:` namespace is machine-written, regenerable
|
|
94
|
+
cache** — never hand-edit, never treat as authoritative;
|
|
95
|
+
* ingestion is an **idempotent full rebuild**: drop all `sys:` memberships,
|
|
96
|
+
re-ingest the stream (no incremental diffing — staleness is the only
|
|
97
|
+
permitted failure mode, drift is not);
|
|
98
|
+
* the projection is **rebuilt locally on each device** from the synced
|
|
99
|
+
SQLite, not synced itself; only the human layer of the DAG is persisted
|
|
100
|
+
and synced (it is the irreplaceable original data — back it up like the
|
|
101
|
+
KeePass vault);
|
|
102
|
+
* human categories are attached to the same hash-identified items, giving
|
|
103
|
+
unified queries like `get(photo, vienna, sys:on:drive-budapest)`.
|
|
104
|
+
|
|
105
|
+
See `ontodag_ingest.py` for an adaptation template.
|
|
106
|
+
|
|
107
|
+
## Notes & limits (v0.1)
|
|
108
|
+
|
|
109
|
+
* `import-restic` matches listing entries to known content by basename+size
|
|
110
|
+
and only when unique; ambiguous entries are recorded as
|
|
111
|
+
`unverified:` placeholders. Scanning a `restic mount` gives exact hashes.
|
|
112
|
+
* Symlinks are skipped. Hidden config/caches are excluded by default
|
|
113
|
+
(`.cache`, `.config`, `.git`, `node_modules`, Syncthing internals, …);
|
|
114
|
+
add your own with `--exclude-file`.
|
|
115
|
+
* Concurrent writes are not supported by design (single-writer model).
|
|
116
|
+
* Roadmap (from the discussion): v0.2 Syncthing REST adapter; v0.3 OntoDAG
|
|
117
|
+
join live; v0.4 FastAPI localhost UI; v0.5 WASM/PWA read-only viewer for
|
|
118
|
+
phones off the synced SQLite.
|
holdings-0.1.0/README.md
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# holdings — v0.1 of the placement catalog
|
|
2
|
+
|
|
3
|
+
A disposable, regenerable catalog answering: **which media hold which files?**
|
|
4
|
+
Single file, stdlib only, Python 3.9+. SQLite for placement truth; semantics
|
|
5
|
+
belong to OntoDAG (see *Projection contract* below).
|
|
6
|
+
|
|
7
|
+
## Design contract (the important part)
|
|
8
|
+
|
|
9
|
+
1. **Observer, not authority.** holdings only *reads* filesystems and backup
|
|
10
|
+
listings. It never writes to your data, never sits in the backup or sync
|
|
11
|
+
path. Deleting holdings and its database costs nothing but convenience.
|
|
12
|
+
2. **Everything is regenerable by re-scanning.** The catalog is a cache of
|
|
13
|
+
facts about the world. The only original data in the whole system is your
|
|
14
|
+
*human* OntoDAG categorization — which holdings never touches.
|
|
15
|
+
3. **Content hash is identity.** `sha256:…` is the primary key everywhere.
|
|
16
|
+
Paths, media, snapshots, categories: all attributes of a hash.
|
|
17
|
+
4. **Single writer, many readers.** Scan on the backup-node laptop. Put the
|
|
18
|
+
SQLite file in a Syncthing folder; every device then carries the full
|
|
19
|
+
index of everything — including drives offline in another country.
|
|
20
|
+
|
|
21
|
+
## Quick start
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
export HOLDINGS_DB=~/Sync/catalog/catalog.sqlite # put it in a synced folder
|
|
25
|
+
|
|
26
|
+
# Register your media once:
|
|
27
|
+
./holdings.py add-medium laptop-x1 --kind laptop --location "with me"
|
|
28
|
+
./holdings.py add-medium drive-budapest --kind drive --backup --location "safe, Budapest"
|
|
29
|
+
./holdings.py add-medium drive-standrews --kind drive --backup --location "office, St Andrews"
|
|
30
|
+
./holdings.py add-medium restic-b2 --kind restic-repo --backup
|
|
31
|
+
|
|
32
|
+
# Scan whenever a medium is mounted (fast on rescan: unchanged files
|
|
33
|
+
# are recognized by size+mtime and not rehashed):
|
|
34
|
+
./holdings.py scan drive-budapest /media/peter/backup-drive
|
|
35
|
+
./holdings.py scan laptop-x1 /home/peter --exclude-file ~/backup/excludes.txt
|
|
36
|
+
|
|
37
|
+
# Count backup snapshots as copies (approximate matching by name+size;
|
|
38
|
+
# for exact hashes, `restic mount` the repo and `scan` it instead):
|
|
39
|
+
restic -r b2:bucket:repo ls --json latest | ./holdings.py import-restic restic-b2 -
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Queries
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
./holdings.py whereis holiday.jpg # every medium+path holding this content
|
|
46
|
+
./holdings.py whereis sha256:45887c... # by hash
|
|
47
|
+
./holdings.py redundancy --min-copies 2 # content below 2 backup copies
|
|
48
|
+
./holdings.py only-on drive-budapest # DANGER LIST: exists nowhere else
|
|
49
|
+
./holdings.py diff drive-a drive-b # on A but not B
|
|
50
|
+
./holdings.py media # media overview
|
|
51
|
+
./holdings.py stats # totals
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
`redundancy` turns your 3-2-1 policy into a checkable report.
|
|
55
|
+
`only-on` is the consolidation to-do list for old scattered drives: run it,
|
|
56
|
+
back those files up via restic, rescan, watch the list empty, then wipe the
|
|
57
|
+
drive with confidence.
|
|
58
|
+
|
|
59
|
+
## Projection contract (OntoDAG integration)
|
|
60
|
+
|
|
61
|
+
*(2026-08-20: this contract's canonical statement now lives at the meet
|
|
62
|
+
point — [ontodag `docs/plans/PROJECTIONS.md`](https://github.com/petfold/ontodag/blob/main/docs/plans/PROJECTIONS.md)
|
|
63
|
+
— which generalizes it across sources (files here, messages in ucomm)
|
|
64
|
+
and adds retention classes. The rules below remain the agreed file-side
|
|
65
|
+
instance and the wire format is unchanged.)*
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
./holdings.py project-ontodag --out placement.jsonl
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Emits JSON lines: `{"item": "<hash>", "supercategories": ["sys:on:<medium>",
|
|
72
|
+
"sys:type:<ext>", "sys:backup:<n>"]}` — matching OntoDAG's
|
|
73
|
+
`put(item, supercategories)` model.
|
|
74
|
+
|
|
75
|
+
The agreed rules for the ingesting side:
|
|
76
|
+
|
|
77
|
+
* everything under the **`sys:` namespace is machine-written, regenerable
|
|
78
|
+
cache** — never hand-edit, never treat as authoritative;
|
|
79
|
+
* ingestion is an **idempotent full rebuild**: drop all `sys:` memberships,
|
|
80
|
+
re-ingest the stream (no incremental diffing — staleness is the only
|
|
81
|
+
permitted failure mode, drift is not);
|
|
82
|
+
* the projection is **rebuilt locally on each device** from the synced
|
|
83
|
+
SQLite, not synced itself; only the human layer of the DAG is persisted
|
|
84
|
+
and synced (it is the irreplaceable original data — back it up like the
|
|
85
|
+
KeePass vault);
|
|
86
|
+
* human categories are attached to the same hash-identified items, giving
|
|
87
|
+
unified queries like `get(photo, vienna, sys:on:drive-budapest)`.
|
|
88
|
+
|
|
89
|
+
See `ontodag_ingest.py` for an adaptation template.
|
|
90
|
+
|
|
91
|
+
## Notes & limits (v0.1)
|
|
92
|
+
|
|
93
|
+
* `import-restic` matches listing entries to known content by basename+size
|
|
94
|
+
and only when unique; ambiguous entries are recorded as
|
|
95
|
+
`unverified:` placeholders. Scanning a `restic mount` gives exact hashes.
|
|
96
|
+
* Symlinks are skipped. Hidden config/caches are excluded by default
|
|
97
|
+
(`.cache`, `.config`, `.git`, `node_modules`, Syncthing internals, …);
|
|
98
|
+
add your own with `--exclude-file`.
|
|
99
|
+
* Concurrent writes are not supported by design (single-writer model).
|
|
100
|
+
* Roadmap (from the discussion): v0.2 Syncthing REST adapter; v0.3 OntoDAG
|
|
101
|
+
join live; v0.4 FastAPI localhost UI; v0.5 WASM/PWA read-only viewer for
|
|
102
|
+
phones off the synced SQLite.
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: holdings
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Disposable, regenerable catalog of which media hold which files
|
|
5
|
+
Author: Peter Foldiak
|
|
6
|
+
License: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/petfold/holdings
|
|
8
|
+
Project-URL: Issues, https://github.com/petfold/holdings/issues
|
|
9
|
+
Keywords: backup,catalog,placement,redundancy,sqlite,content-addressed,ontodag
|
|
10
|
+
Requires-Python: >=3.9
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Provides-Extra: test
|
|
14
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# holdings — v0.1 of the placement catalog
|
|
18
|
+
|
|
19
|
+
A disposable, regenerable catalog answering: **which media hold which files?**
|
|
20
|
+
Single file, stdlib only, Python 3.9+. SQLite for placement truth; semantics
|
|
21
|
+
belong to OntoDAG (see *Projection contract* below).
|
|
22
|
+
|
|
23
|
+
## Design contract (the important part)
|
|
24
|
+
|
|
25
|
+
1. **Observer, not authority.** holdings only *reads* filesystems and backup
|
|
26
|
+
listings. It never writes to your data, never sits in the backup or sync
|
|
27
|
+
path. Deleting holdings and its database costs nothing but convenience.
|
|
28
|
+
2. **Everything is regenerable by re-scanning.** The catalog is a cache of
|
|
29
|
+
facts about the world. The only original data in the whole system is your
|
|
30
|
+
*human* OntoDAG categorization — which holdings never touches.
|
|
31
|
+
3. **Content hash is identity.** `sha256:…` is the primary key everywhere.
|
|
32
|
+
Paths, media, snapshots, categories: all attributes of a hash.
|
|
33
|
+
4. **Single writer, many readers.** Scan on the backup-node laptop. Put the
|
|
34
|
+
SQLite file in a Syncthing folder; every device then carries the full
|
|
35
|
+
index of everything — including drives offline in another country.
|
|
36
|
+
|
|
37
|
+
## Quick start
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
export HOLDINGS_DB=~/Sync/catalog/catalog.sqlite # put it in a synced folder
|
|
41
|
+
|
|
42
|
+
# Register your media once:
|
|
43
|
+
./holdings.py add-medium laptop-x1 --kind laptop --location "with me"
|
|
44
|
+
./holdings.py add-medium drive-budapest --kind drive --backup --location "safe, Budapest"
|
|
45
|
+
./holdings.py add-medium drive-standrews --kind drive --backup --location "office, St Andrews"
|
|
46
|
+
./holdings.py add-medium restic-b2 --kind restic-repo --backup
|
|
47
|
+
|
|
48
|
+
# Scan whenever a medium is mounted (fast on rescan: unchanged files
|
|
49
|
+
# are recognized by size+mtime and not rehashed):
|
|
50
|
+
./holdings.py scan drive-budapest /media/peter/backup-drive
|
|
51
|
+
./holdings.py scan laptop-x1 /home/peter --exclude-file ~/backup/excludes.txt
|
|
52
|
+
|
|
53
|
+
# Count backup snapshots as copies (approximate matching by name+size;
|
|
54
|
+
# for exact hashes, `restic mount` the repo and `scan` it instead):
|
|
55
|
+
restic -r b2:bucket:repo ls --json latest | ./holdings.py import-restic restic-b2 -
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Queries
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
./holdings.py whereis holiday.jpg # every medium+path holding this content
|
|
62
|
+
./holdings.py whereis sha256:45887c... # by hash
|
|
63
|
+
./holdings.py redundancy --min-copies 2 # content below 2 backup copies
|
|
64
|
+
./holdings.py only-on drive-budapest # DANGER LIST: exists nowhere else
|
|
65
|
+
./holdings.py diff drive-a drive-b # on A but not B
|
|
66
|
+
./holdings.py media # media overview
|
|
67
|
+
./holdings.py stats # totals
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
`redundancy` turns your 3-2-1 policy into a checkable report.
|
|
71
|
+
`only-on` is the consolidation to-do list for old scattered drives: run it,
|
|
72
|
+
back those files up via restic, rescan, watch the list empty, then wipe the
|
|
73
|
+
drive with confidence.
|
|
74
|
+
|
|
75
|
+
## Projection contract (OntoDAG integration)
|
|
76
|
+
|
|
77
|
+
*(2026-08-20: this contract's canonical statement now lives at the meet
|
|
78
|
+
point — [ontodag `docs/plans/PROJECTIONS.md`](https://github.com/petfold/ontodag/blob/main/docs/plans/PROJECTIONS.md)
|
|
79
|
+
— which generalizes it across sources (files here, messages in ucomm)
|
|
80
|
+
and adds retention classes. The rules below remain the agreed file-side
|
|
81
|
+
instance and the wire format is unchanged.)*
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
./holdings.py project-ontodag --out placement.jsonl
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Emits JSON lines: `{"item": "<hash>", "supercategories": ["sys:on:<medium>",
|
|
88
|
+
"sys:type:<ext>", "sys:backup:<n>"]}` — matching OntoDAG's
|
|
89
|
+
`put(item, supercategories)` model.
|
|
90
|
+
|
|
91
|
+
The agreed rules for the ingesting side:
|
|
92
|
+
|
|
93
|
+
* everything under the **`sys:` namespace is machine-written, regenerable
|
|
94
|
+
cache** — never hand-edit, never treat as authoritative;
|
|
95
|
+
* ingestion is an **idempotent full rebuild**: drop all `sys:` memberships,
|
|
96
|
+
re-ingest the stream (no incremental diffing — staleness is the only
|
|
97
|
+
permitted failure mode, drift is not);
|
|
98
|
+
* the projection is **rebuilt locally on each device** from the synced
|
|
99
|
+
SQLite, not synced itself; only the human layer of the DAG is persisted
|
|
100
|
+
and synced (it is the irreplaceable original data — back it up like the
|
|
101
|
+
KeePass vault);
|
|
102
|
+
* human categories are attached to the same hash-identified items, giving
|
|
103
|
+
unified queries like `get(photo, vienna, sys:on:drive-budapest)`.
|
|
104
|
+
|
|
105
|
+
See `ontodag_ingest.py` for an adaptation template.
|
|
106
|
+
|
|
107
|
+
## Notes & limits (v0.1)
|
|
108
|
+
|
|
109
|
+
* `import-restic` matches listing entries to known content by basename+size
|
|
110
|
+
and only when unique; ambiguous entries are recorded as
|
|
111
|
+
`unverified:` placeholders. Scanning a `restic mount` gives exact hashes.
|
|
112
|
+
* Symlinks are skipped. Hidden config/caches are excluded by default
|
|
113
|
+
(`.cache`, `.config`, `.git`, `node_modules`, Syncthing internals, …);
|
|
114
|
+
add your own with `--exclude-file`.
|
|
115
|
+
* Concurrent writes are not supported by design (single-writer model).
|
|
116
|
+
* Roadmap (from the discussion): v0.2 Syncthing REST adapter; v0.3 OntoDAG
|
|
117
|
+
join live; v0.4 FastAPI localhost UI; v0.5 WASM/PWA read-only viewer for
|
|
118
|
+
phones off the synced SQLite.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
holdings
|
|
@@ -0,0 +1,587 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
holdings — a disposable, regenerable catalog of where your files are.
|
|
4
|
+
|
|
5
|
+
Design principles (see the accompanying README):
|
|
6
|
+
* Observer, not authority: reads filesystems and backup listings,
|
|
7
|
+
never writes to them. Deleting this tool and its database costs
|
|
8
|
+
nothing but convenience — everything is regenerable by re-scanning.
|
|
9
|
+
* Content hash (sha256) is the primary key. Paths, media, and
|
|
10
|
+
categories are attributes of a hash, never identities.
|
|
11
|
+
* Single writer (your backup-node laptop), many readers (the SQLite
|
|
12
|
+
file can be placed in a Syncthing folder and read anywhere).
|
|
13
|
+
* Placement truth lives here, in SQLite. Semantic truth lives in
|
|
14
|
+
OntoDAG. The `project-ontodag` command emits placement facts as a
|
|
15
|
+
namespaced (`sys:`) projection for OntoDAG to ingest as a
|
|
16
|
+
regenerable, non-authoritative view.
|
|
17
|
+
|
|
18
|
+
Stdlib only. Python 3.9+.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import fnmatch
|
|
23
|
+
import hashlib
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
import sqlite3
|
|
27
|
+
import sys
|
|
28
|
+
import time
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
# --------------------------------------------------------------------------
|
|
32
|
+
# Database
|
|
33
|
+
# --------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
DEFAULT_DB = os.environ.get(
|
|
36
|
+
"HOLDINGS_DB",
|
|
37
|
+
os.path.join(os.path.expanduser("~"), ".local", "share", "holdings",
|
|
38
|
+
"catalog.sqlite"),
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
SCHEMA = """
|
|
42
|
+
CREATE TABLE IF NOT EXISTS media (
|
|
43
|
+
medium_id TEXT PRIMARY KEY, -- e.g. 'drive-budapest', 'laptop-x1', 'restic-b2'
|
|
44
|
+
kind TEXT NOT NULL, -- drive | laptop | phone | restic-repo | cloud | other
|
|
45
|
+
location_hint TEXT, -- e.g. 'safe, Budapest flat'
|
|
46
|
+
is_backup INTEGER NOT NULL DEFAULT 0,-- counts toward redundancy as a backup copy
|
|
47
|
+
notes TEXT,
|
|
48
|
+
last_scanned REAL
|
|
49
|
+
);
|
|
50
|
+
|
|
51
|
+
CREATE TABLE IF NOT EXISTS content (
|
|
52
|
+
hash TEXT PRIMARY KEY, -- 'sha256:...'
|
|
53
|
+
size INTEGER NOT NULL,
|
|
54
|
+
first_seen REAL NOT NULL
|
|
55
|
+
);
|
|
56
|
+
|
|
57
|
+
CREATE TABLE IF NOT EXISTS instances (
|
|
58
|
+
medium_id TEXT NOT NULL REFERENCES media(medium_id),
|
|
59
|
+
path TEXT NOT NULL, -- path relative to the medium's root
|
|
60
|
+
hash TEXT NOT NULL REFERENCES content(hash),
|
|
61
|
+
size INTEGER NOT NULL,
|
|
62
|
+
mtime REAL, -- NULL for imported (e.g. restic) listings
|
|
63
|
+
seen_at REAL NOT NULL,
|
|
64
|
+
PRIMARY KEY (medium_id, path)
|
|
65
|
+
);
|
|
66
|
+
CREATE INDEX IF NOT EXISTS idx_instances_hash ON instances(hash);
|
|
67
|
+
|
|
68
|
+
CREATE TABLE IF NOT EXISTS scans (
|
|
69
|
+
scan_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
70
|
+
medium_id TEXT NOT NULL,
|
|
71
|
+
root TEXT NOT NULL, -- subtree scanned, '' = whole medium
|
|
72
|
+
started REAL NOT NULL,
|
|
73
|
+
finished REAL,
|
|
74
|
+
files_seen INTEGER,
|
|
75
|
+
bytes_seen INTEGER,
|
|
76
|
+
hashed INTEGER -- how many actually (re)hashed
|
|
77
|
+
);
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def db_connect(db_path: str) -> sqlite3.Connection:
|
|
82
|
+
Path(db_path).parent.mkdir(parents=True, exist_ok=True)
|
|
83
|
+
conn = sqlite3.connect(db_path)
|
|
84
|
+
conn.execute("PRAGMA journal_mode=WAL")
|
|
85
|
+
conn.execute("PRAGMA foreign_keys=ON")
|
|
86
|
+
conn.executescript(SCHEMA)
|
|
87
|
+
return conn
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# --------------------------------------------------------------------------
|
|
91
|
+
# Hashing
|
|
92
|
+
# --------------------------------------------------------------------------
|
|
93
|
+
|
|
94
|
+
def sha256_file(path: Path, bufsize: int = 1 << 20) -> str:
|
|
95
|
+
h = hashlib.sha256()
|
|
96
|
+
with open(path, "rb") as f:
|
|
97
|
+
while True:
|
|
98
|
+
chunk = f.read(bufsize)
|
|
99
|
+
if not chunk:
|
|
100
|
+
break
|
|
101
|
+
h.update(chunk)
|
|
102
|
+
return "sha256:" + h.hexdigest()
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# --------------------------------------------------------------------------
|
|
106
|
+
# Excludes
|
|
107
|
+
# --------------------------------------------------------------------------
|
|
108
|
+
|
|
109
|
+
DEFAULT_EXCLUDES = [
|
|
110
|
+
".cache", ".config", ".git", "node_modules", ".Trash-*", ".stfolder",
|
|
111
|
+
".stversions", "lost+found", "*.tmp", ".DS_Store",
|
|
112
|
+
]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def load_excludes(exclude_file: str | None) -> list[str]:
|
|
116
|
+
patterns = list(DEFAULT_EXCLUDES)
|
|
117
|
+
if exclude_file:
|
|
118
|
+
with open(exclude_file) as f:
|
|
119
|
+
for line in f:
|
|
120
|
+
line = line.strip()
|
|
121
|
+
if line and not line.startswith("#"):
|
|
122
|
+
patterns.append(line)
|
|
123
|
+
return patterns
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def is_excluded(rel_path: str, name: str, patterns: list[str]) -> bool:
|
|
127
|
+
for pat in patterns:
|
|
128
|
+
if fnmatch.fnmatch(name, pat) or fnmatch.fnmatch(rel_path, pat):
|
|
129
|
+
return True
|
|
130
|
+
return False
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# --------------------------------------------------------------------------
|
|
134
|
+
# Commands
|
|
135
|
+
# --------------------------------------------------------------------------
|
|
136
|
+
|
|
137
|
+
def cmd_add_medium(conn, args):
|
|
138
|
+
conn.execute(
|
|
139
|
+
"INSERT OR REPLACE INTO media"
|
|
140
|
+
" (medium_id, kind, location_hint, is_backup, notes, last_scanned)"
|
|
141
|
+
" VALUES (?,?,?,?,?,"
|
|
142
|
+
" (SELECT last_scanned FROM media WHERE medium_id=?))",
|
|
143
|
+
(args.medium_id, args.kind, args.location, int(args.backup),
|
|
144
|
+
args.notes, args.medium_id),
|
|
145
|
+
)
|
|
146
|
+
conn.commit()
|
|
147
|
+
print(f"medium '{args.medium_id}' registered"
|
|
148
|
+
f" (kind={args.kind}, backup={'yes' if args.backup else 'no'})")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def cmd_media(conn, args):
|
|
152
|
+
rows = conn.execute(
|
|
153
|
+
"SELECT m.medium_id, m.kind, m.is_backup, m.location_hint,"
|
|
154
|
+
" m.last_scanned,"
|
|
155
|
+
" (SELECT COUNT(*) FROM instances i WHERE i.medium_id=m.medium_id),"
|
|
156
|
+
" (SELECT COALESCE(SUM(size),0) FROM instances i"
|
|
157
|
+
" WHERE i.medium_id=m.medium_id)"
|
|
158
|
+
" FROM media m ORDER BY m.medium_id").fetchall()
|
|
159
|
+
if not rows:
|
|
160
|
+
print("no media registered yet — use: holdings add-medium <id> --kind drive")
|
|
161
|
+
return
|
|
162
|
+
print(f"{'MEDIUM':22} {'KIND':12} {'BK':3} {'FILES':>8} {'SIZE':>10}"
|
|
163
|
+
f" {'LAST SCAN':19} LOCATION")
|
|
164
|
+
for mid, kind, bk, loc, ts, nfiles, nbytes in rows:
|
|
165
|
+
when = time.strftime("%Y-%m-%d %H:%M", time.localtime(ts)) if ts else "never"
|
|
166
|
+
print(f"{mid:22} {kind:12} {'y' if bk else '-':3} {nfiles:>8}"
|
|
167
|
+
f" {human_size(nbytes):>10} {when:19} {loc or ''}")
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def cmd_scan(conn, args):
|
|
171
|
+
medium = conn.execute("SELECT medium_id FROM media WHERE medium_id=?",
|
|
172
|
+
(args.medium_id,)).fetchone()
|
|
173
|
+
if not medium:
|
|
174
|
+
sys.exit(f"unknown medium '{args.medium_id}' — register it first with add-medium")
|
|
175
|
+
|
|
176
|
+
mount = Path(args.mount_path).resolve()
|
|
177
|
+
if not mount.is_dir():
|
|
178
|
+
sys.exit(f"mount path {mount} is not a directory")
|
|
179
|
+
|
|
180
|
+
patterns = load_excludes(args.exclude_file)
|
|
181
|
+
started = time.time()
|
|
182
|
+
root_rel = args.root.strip("/") if args.root else ""
|
|
183
|
+
scan_root = mount / root_rel if root_rel else mount
|
|
184
|
+
|
|
185
|
+
cur = conn.execute(
|
|
186
|
+
"INSERT INTO scans (medium_id, root, started) VALUES (?,?,?)",
|
|
187
|
+
(args.medium_id, root_rel, started))
|
|
188
|
+
scan_id = cur.lastrowid
|
|
189
|
+
|
|
190
|
+
files_seen = bytes_seen = hashed = 0
|
|
191
|
+
for dirpath, dirnames, filenames in os.walk(scan_root):
|
|
192
|
+
rel_dir = os.path.relpath(dirpath, mount)
|
|
193
|
+
rel_dir = "" if rel_dir == "." else rel_dir
|
|
194
|
+
# prune excluded directories in-place
|
|
195
|
+
dirnames[:] = [d for d in dirnames
|
|
196
|
+
if not is_excluded(os.path.join(rel_dir, d), d, patterns)]
|
|
197
|
+
for name in filenames:
|
|
198
|
+
rel_path = os.path.join(rel_dir, name) if rel_dir else name
|
|
199
|
+
if is_excluded(rel_path, name, patterns):
|
|
200
|
+
continue
|
|
201
|
+
full = Path(dirpath) / name
|
|
202
|
+
try:
|
|
203
|
+
st = full.stat()
|
|
204
|
+
except OSError:
|
|
205
|
+
continue
|
|
206
|
+
if not full.is_file() or full.is_symlink():
|
|
207
|
+
continue
|
|
208
|
+
files_seen += 1
|
|
209
|
+
bytes_seen += st.st_size
|
|
210
|
+
|
|
211
|
+
row = conn.execute(
|
|
212
|
+
"SELECT hash, size, mtime FROM instances"
|
|
213
|
+
" WHERE medium_id=? AND path=?",
|
|
214
|
+
(args.medium_id, rel_path)).fetchone()
|
|
215
|
+
if (row and not args.full and row[1] == st.st_size
|
|
216
|
+
and row[2] is not None and abs(row[2] - st.st_mtime) < 1e-6):
|
|
217
|
+
file_hash = row[0] # unchanged: reuse known hash
|
|
218
|
+
else:
|
|
219
|
+
try:
|
|
220
|
+
file_hash = sha256_file(full)
|
|
221
|
+
except OSError as e:
|
|
222
|
+
print(f" ! cannot read {rel_path}: {e}", file=sys.stderr)
|
|
223
|
+
continue
|
|
224
|
+
hashed += 1
|
|
225
|
+
|
|
226
|
+
conn.execute(
|
|
227
|
+
"INSERT INTO content (hash, size, first_seen) VALUES (?,?,?)"
|
|
228
|
+
" ON CONFLICT(hash) DO NOTHING",
|
|
229
|
+
(file_hash, st.st_size, time.time()))
|
|
230
|
+
conn.execute(
|
|
231
|
+
"INSERT INTO instances (medium_id, path, hash, size, mtime, seen_at)"
|
|
232
|
+
" VALUES (?,?,?,?,?,?)"
|
|
233
|
+
" ON CONFLICT(medium_id, path) DO UPDATE SET"
|
|
234
|
+
" hash=excluded.hash, size=excluded.size,"
|
|
235
|
+
" mtime=excluded.mtime, seen_at=excluded.seen_at",
|
|
236
|
+
(args.medium_id, rel_path, file_hash, st.st_size,
|
|
237
|
+
st.st_mtime, time.time()))
|
|
238
|
+
if files_seen % 500 == 0:
|
|
239
|
+
conn.commit()
|
|
240
|
+
print(f" … {files_seen} files ({human_size(bytes_seen)})",
|
|
241
|
+
file=sys.stderr)
|
|
242
|
+
|
|
243
|
+
# prune instances under the scanned root that were not seen this scan
|
|
244
|
+
prefix = (root_rel + "/") if root_rel else ""
|
|
245
|
+
pruned = conn.execute(
|
|
246
|
+
"DELETE FROM instances WHERE medium_id=? AND seen_at<?"
|
|
247
|
+
" AND (path LIKE ? OR ?='')",
|
|
248
|
+
(args.medium_id, started, prefix + "%", prefix)).rowcount
|
|
249
|
+
|
|
250
|
+
conn.execute("UPDATE media SET last_scanned=? WHERE medium_id=?",
|
|
251
|
+
(time.time(), args.medium_id))
|
|
252
|
+
conn.execute(
|
|
253
|
+
"UPDATE scans SET finished=?, files_seen=?, bytes_seen=?, hashed=?"
|
|
254
|
+
" WHERE scan_id=?",
|
|
255
|
+
(time.time(), files_seen, bytes_seen, hashed, scan_id))
|
|
256
|
+
conn.commit()
|
|
257
|
+
print(f"scan of '{args.medium_id}' complete: {files_seen} files"
|
|
258
|
+
f" ({human_size(bytes_seen)}), {hashed} hashed,"
|
|
259
|
+
f" {pruned} vanished entries pruned,"
|
|
260
|
+
f" {time.time()-started:.1f}s")
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def cmd_import_restic(conn, args):
|
|
264
|
+
"""Ingest `restic ls --json <snapshot>` output so backup copies count.
|
|
265
|
+
|
|
266
|
+
Usage:
|
|
267
|
+
restic -r <repo> ls --json latest | holdings import-restic restic-b2 -
|
|
268
|
+
or with a saved file:
|
|
269
|
+
holdings import-restic restic-b2 listing.json
|
|
270
|
+
"""
|
|
271
|
+
medium = conn.execute("SELECT medium_id FROM media WHERE medium_id=?",
|
|
272
|
+
(args.medium_id,)).fetchone()
|
|
273
|
+
if not medium:
|
|
274
|
+
sys.exit(f"unknown medium '{args.medium_id}' — register it first"
|
|
275
|
+
f" (kind=restic-repo, --backup)")
|
|
276
|
+
|
|
277
|
+
stream = sys.stdin if args.listing == "-" else open(args.listing)
|
|
278
|
+
started = time.time()
|
|
279
|
+
count = 0
|
|
280
|
+
with stream:
|
|
281
|
+
for line in stream:
|
|
282
|
+
line = line.strip()
|
|
283
|
+
if not line:
|
|
284
|
+
continue
|
|
285
|
+
try:
|
|
286
|
+
obj = json.loads(line)
|
|
287
|
+
except json.JSONDecodeError:
|
|
288
|
+
continue
|
|
289
|
+
if obj.get("type") != "file" or obj.get("struct_type") == "snapshot":
|
|
290
|
+
continue
|
|
291
|
+
path = obj.get("path", "").lstrip("/")
|
|
292
|
+
size = obj.get("size", 0)
|
|
293
|
+
# restic ls --json does not expose content hashes; identity is
|
|
294
|
+
# matched by (size, name) against known content when unique,
|
|
295
|
+
# otherwise recorded as unverified. For exact matching, scan the
|
|
296
|
+
# restored/mounted repo (restic mount) with `holdings scan`.
|
|
297
|
+
h = match_known_content(conn, path, size)
|
|
298
|
+
if h is None:
|
|
299
|
+
h = f"unverified:restic:{args.medium_id}:{path}:{size}"
|
|
300
|
+
conn.execute(
|
|
301
|
+
"INSERT INTO content (hash, size, first_seen) VALUES (?,?,?)"
|
|
302
|
+
" ON CONFLICT(hash) DO NOTHING", (h, size, time.time()))
|
|
303
|
+
conn.execute(
|
|
304
|
+
"INSERT INTO instances (medium_id, path, hash, size, mtime, seen_at)"
|
|
305
|
+
" VALUES (?,?,?,?,NULL,?)"
|
|
306
|
+
" ON CONFLICT(medium_id, path) DO UPDATE SET"
|
|
307
|
+
" hash=excluded.hash, size=excluded.size, seen_at=excluded.seen_at",
|
|
308
|
+
(args.medium_id, path, h, size, time.time()))
|
|
309
|
+
count += 1
|
|
310
|
+
conn.execute(
|
|
311
|
+
"DELETE FROM instances WHERE medium_id=? AND seen_at<?",
|
|
312
|
+
(args.medium_id, started))
|
|
313
|
+
conn.execute("UPDATE media SET last_scanned=? WHERE medium_id=?",
|
|
314
|
+
(time.time(), args.medium_id))
|
|
315
|
+
conn.commit()
|
|
316
|
+
print(f"imported {count} entries into '{args.medium_id}'"
|
|
317
|
+
f" (exact hashes where filename+size uniquely matched known content;"
|
|
318
|
+
f" run `restic mount` + `holdings scan` for exact verification)")
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def match_known_content(conn, path: str, size: int):
|
|
322
|
+
"""Match an imported listing entry to known content by basename+size,
|
|
323
|
+
only when the match is unique. Conservative: ambiguous → None."""
|
|
324
|
+
base = os.path.basename(path)
|
|
325
|
+
rows = conn.execute(
|
|
326
|
+
"SELECT DISTINCT i.hash FROM instances i"
|
|
327
|
+
" WHERE i.size=? AND (i.path=? OR i.path LIKE ?)"
|
|
328
|
+
" AND i.hash LIKE 'sha256:%' LIMIT 2",
|
|
329
|
+
(size, path, "%/" + base)).fetchall()
|
|
330
|
+
if len(rows) == 1:
|
|
331
|
+
return rows[0][0]
|
|
332
|
+
return None
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def cmd_whereis(conn, args):
|
|
336
|
+
h = resolve_hash(conn, args.target)
|
|
337
|
+
if h is None:
|
|
338
|
+
sys.exit(f"'{args.target}' not found (give a path on a scanned medium,"
|
|
339
|
+
f" a filename, or a sha256:... hash)")
|
|
340
|
+
rows = conn.execute(
|
|
341
|
+
"SELECT i.medium_id, m.kind, m.is_backup, i.path, i.seen_at,"
|
|
342
|
+
" m.location_hint"
|
|
343
|
+
" FROM instances i JOIN media m ON m.medium_id=i.medium_id"
|
|
344
|
+
" WHERE i.hash=? ORDER BY m.is_backup DESC, i.medium_id", (h,)).fetchall()
|
|
345
|
+
size = conn.execute("SELECT size FROM content WHERE hash=?", (h,)).fetchone()
|
|
346
|
+
copies = len({r[0] for r in rows})
|
|
347
|
+
backups = len({r[0] for r in rows if r[2]})
|
|
348
|
+
print(f"{h} ({human_size(size[0]) if size else '?'})")
|
|
349
|
+
print(f"present on {copies} media ({backups} backup):")
|
|
350
|
+
for mid, kind, bk, path, seen, loc in rows:
|
|
351
|
+
when = time.strftime("%Y-%m-%d", time.localtime(seen))
|
|
352
|
+
tag = "backup" if bk else kind
|
|
353
|
+
print(f" [{tag:11}] {mid:20} {path} (seen {when}"
|
|
354
|
+
f"{', ' + loc if loc else ''})")
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def resolve_hash(conn, target: str):
|
|
358
|
+
if target.startswith(("sha256:", "unverified:")):
|
|
359
|
+
return target
|
|
360
|
+
p = Path(target)
|
|
361
|
+
if p.is_file():
|
|
362
|
+
return sha256_file(p)
|
|
363
|
+
# try exact relative path, then basename match
|
|
364
|
+
row = conn.execute("SELECT hash FROM instances WHERE path=? LIMIT 1",
|
|
365
|
+
(target.lstrip("/"),)).fetchone()
|
|
366
|
+
if row:
|
|
367
|
+
return row[0]
|
|
368
|
+
rows = conn.execute(
|
|
369
|
+
"SELECT DISTINCT hash FROM instances WHERE path LIKE ? LIMIT 2",
|
|
370
|
+
("%/" + target,)).fetchall()
|
|
371
|
+
if len(rows) == 1:
|
|
372
|
+
return rows[0][0]
|
|
373
|
+
if len(rows) > 1:
|
|
374
|
+
sys.exit(f"'{target}' is ambiguous — several distinct files share that"
|
|
375
|
+
f" name; give a full path or a hash")
|
|
376
|
+
return None
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def cmd_redundancy(conn, args):
|
|
380
|
+
"""Content with fewer than --min-copies copies on *backup* media."""
|
|
381
|
+
rows = conn.execute(
|
|
382
|
+
"SELECT * FROM ("
|
|
383
|
+
" SELECT c.hash, c.size,"
|
|
384
|
+
" (SELECT COUNT(DISTINCT i.medium_id) FROM instances i"
|
|
385
|
+
" JOIN media m ON m.medium_id=i.medium_id"
|
|
386
|
+
" WHERE i.hash=c.hash AND m.is_backup=1) AS bcopies,"
|
|
387
|
+
" (SELECT COUNT(DISTINCT i2.medium_id) FROM instances i2"
|
|
388
|
+
" WHERE i2.hash=c.hash) AS copies,"
|
|
389
|
+
" (SELECT i3.path FROM instances i3 WHERE i3.hash=c.hash LIMIT 1)"
|
|
390
|
+
" AS example_path"
|
|
391
|
+
" FROM content c"
|
|
392
|
+
") WHERE bcopies < ?"
|
|
393
|
+
" ORDER BY bcopies, size DESC LIMIT ?",
|
|
394
|
+
(args.min_copies, args.limit)).fetchall()
|
|
395
|
+
if not rows:
|
|
396
|
+
print(f"OK: everything has at least {args.min_copies}"
|
|
397
|
+
f" backup cop{'y' if args.min_copies==1 else 'ies'}.")
|
|
398
|
+
return
|
|
399
|
+
total = conn.execute(
|
|
400
|
+
"SELECT COUNT(*) FROM content c WHERE"
|
|
401
|
+
" (SELECT COUNT(DISTINCT i.medium_id) FROM instances i"
|
|
402
|
+
" JOIN media m ON m.medium_id=i.medium_id"
|
|
403
|
+
" WHERE i.hash=c.hash AND m.is_backup=1) < ?",
|
|
404
|
+
(args.min_copies,)).fetchone()[0]
|
|
405
|
+
print(f"{total} content objects below {args.min_copies} backup copies"
|
|
406
|
+
f" (showing up to {args.limit}, largest first):")
|
|
407
|
+
print(f"{'BK':>2} {'ALL':>3} {'SIZE':>10} EXAMPLE PATH")
|
|
408
|
+
for h, size, bcopies, copies, path in rows:
|
|
409
|
+
print(f"{bcopies:>2} {copies:>3} {human_size(size):>10} {path}")
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def cmd_diff(conn, args):
|
|
413
|
+
rows = conn.execute(
|
|
414
|
+
"SELECT i.path, c.size FROM instances i JOIN content c ON c.hash=i.hash"
|
|
415
|
+
" WHERE i.medium_id=? AND i.hash NOT IN"
|
|
416
|
+
" (SELECT hash FROM instances WHERE medium_id=?)"
|
|
417
|
+
" ORDER BY c.size DESC LIMIT ?",
|
|
418
|
+
(args.medium_a, args.medium_b, args.limit)).fetchall()
|
|
419
|
+
total, tbytes = conn.execute(
|
|
420
|
+
"SELECT COUNT(*), COALESCE(SUM(c.size),0) FROM instances i"
|
|
421
|
+
" JOIN content c ON c.hash=i.hash"
|
|
422
|
+
" WHERE i.medium_id=? AND i.hash NOT IN"
|
|
423
|
+
" (SELECT hash FROM instances WHERE medium_id=?)",
|
|
424
|
+
(args.medium_a, args.medium_b)).fetchone()
|
|
425
|
+
print(f"{total} files ({human_size(tbytes)}) on '{args.medium_a}'"
|
|
426
|
+
f" but not on '{args.medium_b}':")
|
|
427
|
+
for path, size in rows:
|
|
428
|
+
print(f" {human_size(size):>10} {path}")
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def cmd_only_on(conn, args):
|
|
432
|
+
"""Content whose ONLY copies are on the given medium — the danger list."""
|
|
433
|
+
rows = conn.execute(
|
|
434
|
+
"SELECT i.path, c.size FROM instances i JOIN content c ON c.hash=i.hash"
|
|
435
|
+
" WHERE i.medium_id=? AND NOT EXISTS"
|
|
436
|
+
" (SELECT 1 FROM instances i2 WHERE i2.hash=i.hash"
|
|
437
|
+
" AND i2.medium_id<>?)"
|
|
438
|
+
" ORDER BY c.size DESC LIMIT ?",
|
|
439
|
+
(args.medium_id, args.medium_id, args.limit)).fetchall()
|
|
440
|
+
total, tbytes = conn.execute(
|
|
441
|
+
"SELECT COUNT(*), COALESCE(SUM(c.size),0) FROM instances i"
|
|
442
|
+
" JOIN content c ON c.hash=i.hash"
|
|
443
|
+
" WHERE i.medium_id=? AND NOT EXISTS"
|
|
444
|
+
" (SELECT 1 FROM instances i2 WHERE i2.hash=i.hash"
|
|
445
|
+
" AND i2.medium_id<>?)",
|
|
446
|
+
(args.medium_id, args.medium_id)).fetchone()
|
|
447
|
+
print(f"{total} files ({human_size(tbytes)}) exist ONLY on"
|
|
448
|
+
f" '{args.medium_id}':")
|
|
449
|
+
for path, size in rows:
|
|
450
|
+
print(f" {human_size(size):>10} {path}")
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def cmd_stats(conn, args):
|
|
454
|
+
n_content, t_bytes = conn.execute(
|
|
455
|
+
"SELECT COUNT(*), COALESCE(SUM(size),0) FROM content").fetchone()
|
|
456
|
+
n_inst = conn.execute("SELECT COUNT(*) FROM instances").fetchone()[0]
|
|
457
|
+
n_media = conn.execute("SELECT COUNT(*) FROM media").fetchone()[0]
|
|
458
|
+
dup = n_inst - n_content if n_content else 0
|
|
459
|
+
print(f"media: {n_media} unique content: {n_content}"
|
|
460
|
+
f" ({human_size(t_bytes)}) instances: {n_inst}"
|
|
461
|
+
f" (dedup would fold {dup} duplicate placements)")
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def cmd_project_ontodag(conn, args):
|
|
465
|
+
"""Emit the `sys:` placement projection for OntoDAG.
|
|
466
|
+
|
|
467
|
+
Output: JSON lines, one item per content object:
|
|
468
|
+
{"item": "<hash>", "supercategories": ["sys:on:<medium>", ...,
|
|
469
|
+
"sys:type:<ext>", "sys:backup:<n>"]}
|
|
470
|
+
Intended contract (per the agreed design):
|
|
471
|
+
* everything under `sys:` is machine-written, regenerable cache;
|
|
472
|
+
* the ingesting side should DROP all `sys:` memberships and rebuild
|
|
473
|
+
from this stream (idempotent full rebuild, no incremental diffing);
|
|
474
|
+
* human categories are never touched by this stream.
|
|
475
|
+
"""
|
|
476
|
+
out = sys.stdout if args.out == "-" else open(args.out, "w")
|
|
477
|
+
rows = conn.execute(
|
|
478
|
+
"SELECT c.hash,"
|
|
479
|
+
" (SELECT GROUP_CONCAT(DISTINCT i.medium_id) FROM instances i"
|
|
480
|
+
" WHERE i.hash=c.hash),"
|
|
481
|
+
" (SELECT COUNT(DISTINCT i2.medium_id) FROM instances i2"
|
|
482
|
+
" JOIN media m ON m.medium_id=i2.medium_id"
|
|
483
|
+
" WHERE i2.hash=c.hash AND m.is_backup=1),"
|
|
484
|
+
" (SELECT i3.path FROM instances i3 WHERE i3.hash=c.hash LIMIT 1)"
|
|
485
|
+
" FROM content c").fetchall()
|
|
486
|
+
n = 0
|
|
487
|
+
with out:
|
|
488
|
+
for h, media_csv, bcopies, path in rows:
|
|
489
|
+
supers = [f"sys:on:{m}" for m in (media_csv or "").split(",") if m]
|
|
490
|
+
ext = os.path.splitext(path or "")[1].lstrip(".").lower()
|
|
491
|
+
if ext:
|
|
492
|
+
supers.append(f"sys:type:{ext}")
|
|
493
|
+
supers.append(f"sys:backup:{bcopies}")
|
|
494
|
+
out.write(json.dumps({"item": h, "supercategories": supers}) + "\n")
|
|
495
|
+
n += 1
|
|
496
|
+
print(f"projected {n} items", file=sys.stderr)
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
def human_size(n) -> str:
|
|
500
|
+
n = n or 0
|
|
501
|
+
for unit in ("B", "KB", "MB", "GB", "TB"):
|
|
502
|
+
if n < 1024 or unit == "TB":
|
|
503
|
+
return f"{n:.0f}{unit}" if unit == "B" else f"{n:.1f}{unit}"
|
|
504
|
+
n /= 1024
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
# --------------------------------------------------------------------------
|
|
508
|
+
# CLI wiring
|
|
509
|
+
# --------------------------------------------------------------------------
|
|
510
|
+
|
|
511
|
+
def main(argv=None):
|
|
512
|
+
p = argparse.ArgumentParser(
|
|
513
|
+
prog="holdings",
|
|
514
|
+
description="Catalog of which media hold which files."
|
|
515
|
+
" Placement truth in SQLite; semantics belong to OntoDAG.")
|
|
516
|
+
p.add_argument("--db", default=DEFAULT_DB,
|
|
517
|
+
help=f"catalog database (default {DEFAULT_DB};"
|
|
518
|
+
f" put it in a Syncthing folder to read it everywhere)")
|
|
519
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
520
|
+
|
|
521
|
+
s = sub.add_parser("add-medium", help="register a medium")
|
|
522
|
+
s.add_argument("medium_id")
|
|
523
|
+
s.add_argument("--kind", required=True,
|
|
524
|
+
choices=["drive", "laptop", "phone", "restic-repo",
|
|
525
|
+
"cloud", "other"])
|
|
526
|
+
s.add_argument("--location", help="where it physically lives")
|
|
527
|
+
s.add_argument("--backup", action="store_true",
|
|
528
|
+
help="counts toward backup redundancy")
|
|
529
|
+
s.add_argument("--notes")
|
|
530
|
+
s.set_defaults(func=cmd_add_medium)
|
|
531
|
+
|
|
532
|
+
s = sub.add_parser("media", help="list media")
|
|
533
|
+
s.set_defaults(func=cmd_media)
|
|
534
|
+
|
|
535
|
+
s = sub.add_parser("scan", help="scan a mounted medium (or subtree)")
|
|
536
|
+
s.add_argument("medium_id")
|
|
537
|
+
s.add_argument("mount_path")
|
|
538
|
+
s.add_argument("--root", help="only scan this subtree (relative)")
|
|
539
|
+
s.add_argument("--exclude-file", help="extra exclude patterns, one per line")
|
|
540
|
+
s.add_argument("--full", action="store_true",
|
|
541
|
+
help="rehash everything (ignore mtime+size shortcut)")
|
|
542
|
+
s.set_defaults(func=cmd_scan)
|
|
543
|
+
|
|
544
|
+
s = sub.add_parser("import-restic",
|
|
545
|
+
help="ingest `restic ls --json` output ('-' = stdin)")
|
|
546
|
+
s.add_argument("medium_id")
|
|
547
|
+
s.add_argument("listing")
|
|
548
|
+
s.set_defaults(func=cmd_import_restic)
|
|
549
|
+
|
|
550
|
+
s = sub.add_parser("whereis", help="which media hold this file?")
|
|
551
|
+
s.add_argument("target", help="path, filename, or sha256:... hash")
|
|
552
|
+
s.set_defaults(func=cmd_whereis)
|
|
553
|
+
|
|
554
|
+
s = sub.add_parser("redundancy", help="content below N backup copies")
|
|
555
|
+
s.add_argument("--min-copies", type=int, default=2)
|
|
556
|
+
s.add_argument("--limit", type=int, default=40)
|
|
557
|
+
s.set_defaults(func=cmd_redundancy)
|
|
558
|
+
|
|
559
|
+
s = sub.add_parser("diff", help="on A but not on B")
|
|
560
|
+
s.add_argument("medium_a")
|
|
561
|
+
s.add_argument("medium_b")
|
|
562
|
+
s.add_argument("--limit", type=int, default=40)
|
|
563
|
+
s.set_defaults(func=cmd_diff)
|
|
564
|
+
|
|
565
|
+
s = sub.add_parser("only-on", help="content that exists ONLY on this medium")
|
|
566
|
+
s.add_argument("medium_id")
|
|
567
|
+
s.add_argument("--limit", type=int, default=40)
|
|
568
|
+
s.set_defaults(func=cmd_only_on)
|
|
569
|
+
|
|
570
|
+
s = sub.add_parser("stats", help="catalog totals")
|
|
571
|
+
s.set_defaults(func=cmd_stats)
|
|
572
|
+
|
|
573
|
+
s = sub.add_parser("project-ontodag",
|
|
574
|
+
help="emit sys: placement projection (JSON lines)")
|
|
575
|
+
s.add_argument("--out", default="-")
|
|
576
|
+
s.set_defaults(func=cmd_project_ontodag)
|
|
577
|
+
|
|
578
|
+
args = p.parse_args(argv)
|
|
579
|
+
conn = db_connect(args.db)
|
|
580
|
+
try:
|
|
581
|
+
args.func(conn, args)
|
|
582
|
+
finally:
|
|
583
|
+
conn.close()
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
if __name__ == "__main__":
|
|
587
|
+
main()
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "holdings"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Disposable, regenerable catalog of which media hold which files"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = { text = "BSD-3-Clause" }
|
|
7
|
+
requires-python = ">=3.9"
|
|
8
|
+
authors = [{ name = "Peter Foldiak" }]
|
|
9
|
+
keywords = ["backup", "catalog", "placement", "redundancy", "sqlite", "content-addressed", "ontodag"]
|
|
10
|
+
dependencies = []
|
|
11
|
+
|
|
12
|
+
[project.optional-dependencies]
|
|
13
|
+
test = ["pytest>=8"]
|
|
14
|
+
|
|
15
|
+
[project.urls]
|
|
16
|
+
Homepage = "https://github.com/petfold/holdings"
|
|
17
|
+
Issues = "https://github.com/petfold/holdings/issues"
|
|
18
|
+
|
|
19
|
+
[project.scripts]
|
|
20
|
+
holdings = "holdings:main"
|
|
21
|
+
|
|
22
|
+
[build-system]
|
|
23
|
+
requires = ["setuptools>=68"]
|
|
24
|
+
build-backend = "setuptools.build_meta"
|
|
25
|
+
|
|
26
|
+
[tool.setuptools]
|
|
27
|
+
py-modules = ["holdings"]
|
holdings-0.1.0/setup.cfg
ADDED