@mobileaidev/ai-app-bridge 0.2.15 → 0.3.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/NOTICE +7 -0
- package/README.md +306 -53
- package/bin/ai-app-bridge.js +51 -4504
- package/bin/android-permissions.js +152 -0
- package/bin/android-uia-xml.js +74 -0
- package/bin/artifact-paths.js +127 -1
- package/bin/bridge-forward.js +56 -0
- package/bin/command-discovery.js +86 -0
- package/bin/command-errors.js +60 -0
- package/bin/command-registry.js +504 -0
- package/bin/command-request.js +14 -0
- package/bin/command-router.js +80 -0
- package/bin/connection-cache.js +119 -0
- package/bin/device-provider.js +2934 -0
- package/bin/execution-host.js +596 -0
- package/bin/execution-runtime.js +101 -0
- package/bin/fact-codec.js +321 -0
- package/bin/fact-recorder.js +691 -0
- package/bin/fact-store.js +226 -0
- package/bin/feedback-probe.js +285 -0
- package/bin/intent/install-intent.js +267 -0
- package/bin/intent/intent-action-executor.js +129 -0
- package/bin/intent/intent-autonomous-adapter.js +46 -0
- package/bin/intent/intent-capture-port.js +64 -0
- package/bin/intent/intent-entry.js +226 -0
- package/bin/intent/intent-errors.js +28 -0
- package/bin/intent/intent-evidence-store.js +121 -0
- package/bin/intent/intent-lifetime.js +62 -0
- package/bin/intent/intent-observation-target.js +33 -0
- package/bin/intent/intent-observer.js +194 -0
- package/bin/intent/intent-production-adapter.js +334 -0
- package/bin/intent/intent-provider.js +27 -0
- package/bin/intent/intent-runtime.js +63 -0
- package/bin/intent/intent-worker.js +432 -0
- package/bin/intent/ios-intent-adapter.js +90 -0
- package/bin/intent/permission-intent.js +239 -0
- package/bin/intent/web-intent-adapter.js +31 -0
- package/bin/ios-device-outcome.js +47 -0
- package/bin/ios-execution.js +108 -0
- package/bin/ios-provider.js +497 -583
- package/bin/ios-runtime-binding.js +54 -0
- package/bin/ios-wda-execution.js +70 -0
- package/bin/ios-wda-port.js +98 -0
- package/bin/ios-wda-project.js +79 -0
- package/bin/mcp-server.js +83 -1040
- package/bin/mmap-scan-index.js +329 -0
- package/bin/observation-collector.js +862 -0
- package/bin/runtime-client.js +154 -0
- package/bin/runtime-directory.js +107 -0
- package/bin/runtime-protocol.js +36 -0
- package/bin/script/bounded-script-registry.js +103 -0
- package/bin/script/node-runtime-adapter.js +233 -0
- package/bin/script/progress-projector.js +63 -0
- package/bin/script/python-runtime-adapter.js +111 -0
- package/bin/script/rolling-summary.js +134 -0
- package/bin/script/script-agent-port.js +40 -0
- package/bin/script/script-assert.js +95 -0
- package/bin/script/script-capture-port.js +76 -0
- package/bin/script/script-catalog.js +85 -0
- package/bin/script/script-durable-restore.js +195 -0
- package/bin/script/script-entry-code.js +26 -0
- package/bin/script/script-entry-route.js +38 -0
- package/bin/script/script-entry.js +3 -0
- package/bin/script/script-errors.js +29 -0
- package/bin/script/script-evidence-store.js +22 -0
- package/bin/script/script-format-removed.js +26 -0
- package/bin/script/script-host-port.js +397 -0
- package/bin/script/script-ledger.js +64 -0
- package/bin/script/script-result.js +57 -0
- package/bin/script/script-sdk.js +152 -0
- package/bin/script/script-sdk.py +153 -0
- package/bin/script/script-session-channel.js +127 -0
- package/bin/script/script-spec.js +86 -0
- package/bin/script/script-supervisor.js +919 -0
- package/bin/script/templates/checkpoint-reentry.js +13 -0
- package/bin/segment-index.js +481 -0
- package/bin/segmented-fact-store.js +1571 -0
- package/bin/shared-kernel/android-h5-target.js +10 -0
- package/bin/shared-kernel/android-install-execution.js +176 -0
- package/bin/shared-kernel/android-sdk-endpoint.js +42 -0
- package/bin/shared-kernel/android-shell-execution.js +195 -0
- package/bin/shared-kernel/argument-schema.js +117 -0
- package/bin/shared-kernel/canonical-path.js +17 -0
- package/bin/shared-kernel/device-acknowledgements.js +53 -0
- package/bin/shared-kernel/device-completion-history.js +52 -0
- package/bin/shared-kernel/device-mutation-lease.js +219 -0
- package/bin/shared-kernel/device-ownership-recovery.js +95 -0
- package/bin/shared-kernel/device-ownership-store.js +94 -0
- package/bin/shared-kernel/evidence-adapters.js +251 -0
- package/bin/shared-kernel/evidence-archive.js +329 -0
- package/bin/shared-kernel/evidence-recording.js +131 -0
- package/bin/shared-kernel/evidence-schema.js +194 -0
- package/bin/shared-kernel/evidence-store.js +190 -0
- package/bin/shared-kernel/execution-admission.js +22 -0
- package/bin/shared-kernel/execution-contracts.js +171 -0
- package/bin/shared-kernel/execution-io.js +106 -0
- package/bin/shared-kernel/execution-ledger.js +125 -0
- package/bin/shared-kernel/execution-scope.js +87 -0
- package/bin/shared-kernel/execution-target.js +115 -0
- package/bin/shared-kernel/flutter-execution.js +13 -0
- package/bin/shared-kernel/flutter-h5-port.js +60 -0
- package/bin/shared-kernel/flutter-h5-target.js +9 -0
- package/bin/shared-kernel/flutter-target.js +75 -0
- package/bin/shared-kernel/h5-execution.js +11 -0
- package/bin/shared-kernel/h5-target.js +31 -0
- package/bin/shared-kernel/host-fact-store.js +49 -0
- package/bin/shared-kernel/ios-h5-target.js +9 -0
- package/bin/shared-kernel/ios-native-target.js +71 -0
- package/bin/shared-kernel/live-capture-query.js +115 -0
- package/bin/shared-kernel/managed-sdk-execution.js +78 -0
- package/bin/shared-kernel/native-execution.js +13 -0
- package/bin/shared-kernel/native-target.js +156 -0
- package/bin/shared-kernel/provider-command-contracts.js +55 -0
- package/bin/shared-kernel/recorded-payload-archive.js +195 -0
- package/bin/shared-kernel/request-context.js +35 -0
- package/bin/shared-kernel/semantic-node.js +55 -0
- package/bin/shared-kernel/summary-transformer.js +352 -0
- package/bin/shared-kernel/target-lease-protocol.js +47 -0
- package/bin/shared-kernel/text-wait.js +111 -0
- package/bin/shared-kernel/uia-execution.js +96 -0
- package/bin/shared-kernel/uia-protocol.js +214 -0
- package/bin/shared-kernel/uia-runtime-port.js +377 -0
- package/bin/shared-kernel/uia-target.js +39 -0
- package/bin/shared-kernel/web-dom-target.js +44 -0
- package/bin/shared-kernel/xml-attributes.js +25 -0
- package/bin/target-execution.js +275 -0
- package/bin/web/command-schema.js +60 -0
- package/bin/web/session-store.js +157 -0
- package/bin/web-provider.js +334 -553
- package/docs/COMMAND_CONTRACT.md +1563 -0
- package/docs/EVIDENCE_ARCHIVE.md +214 -0
- package/docs/INTENT_FOREGROUND.md +71 -0
- package/docs/INTENT_NATIVE_EDITING.md +79 -0
- package/docs/RELEASE.md +59 -0
- package/docs/SCRIPT_AUTHORING.md +489 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/LICENSE +201 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/NOTICE +7 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/binding.gyp +36 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/bindings/node/sfs_node.c +597 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/include/sfs.h +178 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/index.js +5 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/package.json +24 -0
- package/node_modules/@mobileaidev/segmented-fact-store-native/src/sfs.c +2349 -0
- package/package.json +60 -5
- package/runtime/ios-wda/AABWDABinding.h +19 -0
- package/runtime/ios-wda/AABWDABinding.m +97 -0
- package/runtime/ios-wda/AABWDAExecution.h +26 -0
- package/runtime/ios-wda/AABWDAExecution.m +172 -0
- package/runtime/ios-wda/AABWDAIntegration.h +71 -0
- package/runtime/ios-wda/AABWDAManagedRoutes.h +392 -0
- package/runtime/ios-wda/AABWDAReceiptStore.h +10 -0
- package/runtime/ios-wda/AABWDAReceiptStore.m +116 -0
- package/runtime/uia/ai-app-bridge-uia.jar +0 -0
- package/runtime/uia/manifest.json +22 -0
- package/skills/ai-app-bridge-use/SKILL.md +23 -360
|
@@ -0,0 +1,2349 @@
|
|
|
1
|
+
#define _DEFAULT_SOURCE
|
|
2
|
+
#define _DARWIN_C_SOURCE
|
|
3
|
+
#define _POSIX_C_SOURCE 200809L
|
|
4
|
+
|
|
5
|
+
#include "sfs.h"
|
|
6
|
+
|
|
7
|
+
#include <dirent.h>
|
|
8
|
+
#include <errno.h>
|
|
9
|
+
#include <fcntl.h>
|
|
10
|
+
#include <inttypes.h>
|
|
11
|
+
#include <pthread.h>
|
|
12
|
+
#include <stdarg.h>
|
|
13
|
+
#include <stdatomic.h>
|
|
14
|
+
#include <stdbool.h>
|
|
15
|
+
#include <stdio.h>
|
|
16
|
+
#include <stdlib.h>
|
|
17
|
+
#include <string.h>
|
|
18
|
+
#include <sys/file.h>
|
|
19
|
+
#include <sys/mman.h>
|
|
20
|
+
#include <sys/stat.h>
|
|
21
|
+
#include <sys/types.h>
|
|
22
|
+
#include <unistd.h>
|
|
23
|
+
|
|
24
|
+
#define SFS_MIN_SEGMENT_SIZE 128u
|
|
25
|
+
#define SFS_FRAME_PREFIX_SIZE 24u
|
|
26
|
+
#define SFS_FRAME_COMMIT_SIZE 8u
|
|
27
|
+
#define SFS_MIN_FRAME_SIZE (SFS_FRAME_PREFIX_SIZE + SFS_FRAME_COMMIT_SIZE)
|
|
28
|
+
|
|
29
|
+
#define SFS_MANIFEST_FILE ".sfs-manifest"
|
|
30
|
+
#define SFS_LOCK_FILE ".sfs-lock"
|
|
31
|
+
#define SFS_MANIFEST_SIZE 4096u
|
|
32
|
+
#define SFS_MANIFEST_SLOT_SIZE 512u
|
|
33
|
+
#define SFS_MANIFEST_SLOT_COUNT 2u
|
|
34
|
+
#define SFS_MANIFEST_PARTITIONS_OFFSET 48u
|
|
35
|
+
#define SFS_MANIFEST_PARTITION_SIZE 48u
|
|
36
|
+
#define SFS_MANIFEST_CRC_OFFSET 432u
|
|
37
|
+
#define SFS_MANIFEST_COMMIT_OFFSET 504u
|
|
38
|
+
|
|
39
|
+
#define SFS_SEGMENT_CRC_OFFSET 48u
|
|
40
|
+
#define SFS_INTERNAL_TORN 100
|
|
41
|
+
|
|
42
|
+
static const uint8_t SFS_MANIFEST_MAGIC[8] = {
|
|
43
|
+
'S', 'F', 'S', 'M', 'A', 'N', '0', '1'};
|
|
44
|
+
static const uint8_t SFS_SEGMENT_MAGIC[8] = {
|
|
45
|
+
'S', 'F', 'S', 'S', 'E', 'G', '0', '1'};
|
|
46
|
+
static const uint64_t SFS_MANIFEST_COMMIT_MARKER = UINT64_C(0x314d4f434d534653);
|
|
47
|
+
static const uint64_t SFS_FRAME_COMMIT_MARKER = UINT64_C(0x314d4f4346534653);
|
|
48
|
+
|
|
49
|
+
typedef struct sfs_segment_meta {
|
|
50
|
+
uint64_t id;
|
|
51
|
+
uint64_t write_offset;
|
|
52
|
+
uint64_t record_count;
|
|
53
|
+
uint64_t payload_bytes;
|
|
54
|
+
uint64_t first_sequence;
|
|
55
|
+
uint64_t last_sequence;
|
|
56
|
+
} sfs_segment_meta_t;
|
|
57
|
+
|
|
58
|
+
typedef struct sfs_partition {
|
|
59
|
+
bool enabled;
|
|
60
|
+
uint64_t quota_bytes;
|
|
61
|
+
char *directory;
|
|
62
|
+
sfs_segment_meta_t *segments;
|
|
63
|
+
size_t segment_count;
|
|
64
|
+
size_t segment_capacity;
|
|
65
|
+
int active_fd;
|
|
66
|
+
uint8_t *active_map;
|
|
67
|
+
uint64_t active_write_offset;
|
|
68
|
+
uint64_t durable_offset;
|
|
69
|
+
uint64_t next_segment_id;
|
|
70
|
+
uint64_t first_retained_segment_id;
|
|
71
|
+
uint64_t record_count;
|
|
72
|
+
uint64_t payload_bytes;
|
|
73
|
+
uint64_t first_sequence;
|
|
74
|
+
uint64_t last_sequence;
|
|
75
|
+
uint64_t evicted_segments;
|
|
76
|
+
uint64_t evicted_records;
|
|
77
|
+
uint64_t evicted_payload_bytes;
|
|
78
|
+
/* Global scans merge partitions by sequence. Retain only the proven read
|
|
79
|
+
position, never payloads or results; rewinds start from segment metadata. */
|
|
80
|
+
uint64_t scan_after_sequence;
|
|
81
|
+
uint64_t scan_segment_id;
|
|
82
|
+
uint64_t scan_offset;
|
|
83
|
+
} sfs_partition_t;
|
|
84
|
+
|
|
85
|
+
struct sfs_store {
|
|
86
|
+
pthread_mutex_t mutex;
|
|
87
|
+
bool mutex_initialized;
|
|
88
|
+
char *directory;
|
|
89
|
+
int lock_fd;
|
|
90
|
+
int manifest_fd;
|
|
91
|
+
uint8_t *manifest_map;
|
|
92
|
+
int manifest_active_slot;
|
|
93
|
+
uint64_t manifest_generation;
|
|
94
|
+
uint64_t segment_size;
|
|
95
|
+
uint64_t next_sequence;
|
|
96
|
+
uint32_t status_flags;
|
|
97
|
+
uint32_t recovery_partition_id;
|
|
98
|
+
uint64_t recovery_segment_id;
|
|
99
|
+
uint64_t recovery_offset;
|
|
100
|
+
uint64_t recovery_discarded_bytes;
|
|
101
|
+
sfs_partition_t partitions[SFS_MAX_PARTITIONS];
|
|
102
|
+
};
|
|
103
|
+
|
|
104
|
+
typedef struct sfs_loaded_manifest_partition {
|
|
105
|
+
uint64_t quota_bytes;
|
|
106
|
+
uint64_t next_segment_id;
|
|
107
|
+
uint64_t first_retained_segment_id;
|
|
108
|
+
uint64_t evicted_segments;
|
|
109
|
+
uint64_t evicted_records;
|
|
110
|
+
uint64_t evicted_payload_bytes;
|
|
111
|
+
} sfs_loaded_manifest_partition_t;
|
|
112
|
+
|
|
113
|
+
typedef struct sfs_loaded_manifest {
|
|
114
|
+
uint64_t generation;
|
|
115
|
+
uint64_t next_sequence;
|
|
116
|
+
uint64_t segment_size;
|
|
117
|
+
sfs_loaded_manifest_partition_t partitions[SFS_MAX_PARTITIONS];
|
|
118
|
+
} sfs_loaded_manifest_t;
|
|
119
|
+
|
|
120
|
+
typedef struct sfs_frame_view {
|
|
121
|
+
uint32_t total_length;
|
|
122
|
+
uint32_t payload_length;
|
|
123
|
+
uint64_t sequence;
|
|
124
|
+
uint64_t payload_offset;
|
|
125
|
+
uint64_t commit_offset;
|
|
126
|
+
} sfs_frame_view_t;
|
|
127
|
+
|
|
128
|
+
typedef struct sfs_candidate {
|
|
129
|
+
bool present;
|
|
130
|
+
uint32_t partition_id;
|
|
131
|
+
size_t segment_index;
|
|
132
|
+
uint64_t frame_offset;
|
|
133
|
+
sfs_frame_view_t frame;
|
|
134
|
+
} sfs_candidate_t;
|
|
135
|
+
|
|
136
|
+
static pthread_once_t sfs_crc_once = PTHREAD_ONCE_INIT;
|
|
137
|
+
static uint32_t sfs_crc_table[256];
|
|
138
|
+
|
|
139
|
+
static void sfs_crc_initialize(void)
|
|
140
|
+
{
|
|
141
|
+
uint32_t index;
|
|
142
|
+
for (index = 0u; index < 256u; ++index) {
|
|
143
|
+
uint32_t value = index;
|
|
144
|
+
uint32_t bit;
|
|
145
|
+
for (bit = 0u; bit < 8u; ++bit) {
|
|
146
|
+
value = (value >> 1u) ^
|
|
147
|
+
((value & 1u) != 0u ? UINT32_C(0x82f63b78) : 0u);
|
|
148
|
+
}
|
|
149
|
+
sfs_crc_table[index] = value;
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
static uint32_t sfs_crc32c(const uint8_t *data, size_t length)
|
|
154
|
+
{
|
|
155
|
+
uint32_t crc = UINT32_MAX;
|
|
156
|
+
size_t index;
|
|
157
|
+
(void)pthread_once(&sfs_crc_once, sfs_crc_initialize);
|
|
158
|
+
for (index = 0u; index < length; ++index) {
|
|
159
|
+
crc = sfs_crc_table[(crc ^ data[index]) & 0xffu] ^ (crc >> 8u);
|
|
160
|
+
}
|
|
161
|
+
return ~crc;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
static uint32_t load_u32_le(const uint8_t *bytes)
|
|
165
|
+
{
|
|
166
|
+
return ((uint32_t)bytes[0]) | ((uint32_t)bytes[1] << 8u) |
|
|
167
|
+
((uint32_t)bytes[2] << 16u) | ((uint32_t)bytes[3] << 24u);
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
static uint64_t load_u64_le(const uint8_t *bytes)
|
|
171
|
+
{
|
|
172
|
+
return ((uint64_t)load_u32_le(bytes)) |
|
|
173
|
+
((uint64_t)load_u32_le(bytes + 4u) << 32u);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
static void store_u32_le(uint8_t *bytes, uint32_t value)
|
|
177
|
+
{
|
|
178
|
+
bytes[0] = (uint8_t)value;
|
|
179
|
+
bytes[1] = (uint8_t)(value >> 8u);
|
|
180
|
+
bytes[2] = (uint8_t)(value >> 16u);
|
|
181
|
+
bytes[3] = (uint8_t)(value >> 24u);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
static void store_u64_le(uint8_t *bytes, uint64_t value)
|
|
185
|
+
{
|
|
186
|
+
store_u32_le(bytes, (uint32_t)value);
|
|
187
|
+
store_u32_le(bytes + 4u, (uint32_t)(value >> 32u));
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
static uint64_t align_eight(uint64_t value)
|
|
191
|
+
{
|
|
192
|
+
return (value + 7u) & ~UINT64_C(7);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
static void clear_error(sfs_error_t *error)
|
|
196
|
+
{
|
|
197
|
+
if (error != NULL) {
|
|
198
|
+
memset(error, 0, sizeof(*error));
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
static sfs_result_t set_error(sfs_error_t *error,
|
|
203
|
+
sfs_result_t result,
|
|
204
|
+
int system_code,
|
|
205
|
+
const char *format,
|
|
206
|
+
...)
|
|
207
|
+
{
|
|
208
|
+
if (error != NULL) {
|
|
209
|
+
va_list arguments;
|
|
210
|
+
memset(error, 0, sizeof(*error));
|
|
211
|
+
error->code = result;
|
|
212
|
+
error->system_code = system_code;
|
|
213
|
+
va_start(arguments, format);
|
|
214
|
+
(void)vsnprintf(error->message, sizeof(error->message), format, arguments);
|
|
215
|
+
va_end(arguments);
|
|
216
|
+
}
|
|
217
|
+
return result;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
static char *path_join(const char *left, const char *right)
|
|
221
|
+
{
|
|
222
|
+
size_t left_length = strlen(left);
|
|
223
|
+
size_t right_length = strlen(right);
|
|
224
|
+
char *path;
|
|
225
|
+
if (left_length > SIZE_MAX - right_length - 2u) {
|
|
226
|
+
return NULL;
|
|
227
|
+
}
|
|
228
|
+
path = (char *)malloc(left_length + right_length + 2u);
|
|
229
|
+
if (path == NULL) {
|
|
230
|
+
return NULL;
|
|
231
|
+
}
|
|
232
|
+
(void)snprintf(path, left_length + right_length + 2u, "%s/%s", left, right);
|
|
233
|
+
return path;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
static char *partition_path(const char *directory, uint32_t partition_id)
|
|
237
|
+
{
|
|
238
|
+
char name[32];
|
|
239
|
+
(void)snprintf(name, sizeof(name), "partition-%u", partition_id);
|
|
240
|
+
return path_join(directory, name);
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
static char *segment_path(const sfs_partition_t *partition, uint64_t segment_id)
|
|
244
|
+
{
|
|
245
|
+
char name[48];
|
|
246
|
+
(void)snprintf(name, sizeof(name), "segment-%020" PRIu64 ".sfs", segment_id);
|
|
247
|
+
return path_join(partition->directory, name);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
static char *temporary_segment_path(const sfs_partition_t *partition,
|
|
251
|
+
uint64_t segment_id)
|
|
252
|
+
{
|
|
253
|
+
char name[56];
|
|
254
|
+
(void)snprintf(name,
|
|
255
|
+
sizeof(name),
|
|
256
|
+
".segment-%020" PRIu64 ".creating",
|
|
257
|
+
segment_id);
|
|
258
|
+
return path_join(partition->directory, name);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
static sfs_result_t ensure_directory(const char *path,
|
|
262
|
+
bool create,
|
|
263
|
+
sfs_error_t *error)
|
|
264
|
+
{
|
|
265
|
+
struct stat status;
|
|
266
|
+
if (stat(path, &status) == 0) {
|
|
267
|
+
if (!S_ISDIR(status.st_mode)) {
|
|
268
|
+
return set_error(error,
|
|
269
|
+
SFS_ERR_IO,
|
|
270
|
+
ENOTDIR,
|
|
271
|
+
"path is not a directory: %s",
|
|
272
|
+
path);
|
|
273
|
+
}
|
|
274
|
+
return SFS_OK;
|
|
275
|
+
}
|
|
276
|
+
if (errno != ENOENT || !create) {
|
|
277
|
+
int saved_errno = errno;
|
|
278
|
+
return set_error(error,
|
|
279
|
+
SFS_ERR_IO,
|
|
280
|
+
saved_errno,
|
|
281
|
+
"cannot access directory %s: %s",
|
|
282
|
+
path,
|
|
283
|
+
strerror(saved_errno));
|
|
284
|
+
}
|
|
285
|
+
if (mkdir(path, 0700) != 0) {
|
|
286
|
+
int saved_errno = errno;
|
|
287
|
+
return set_error(error,
|
|
288
|
+
SFS_ERR_IO,
|
|
289
|
+
saved_errno,
|
|
290
|
+
"cannot create directory %s: %s",
|
|
291
|
+
path,
|
|
292
|
+
strerror(saved_errno));
|
|
293
|
+
}
|
|
294
|
+
return SFS_OK;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
static void fsync_directory_best_effort(const char *path)
|
|
298
|
+
{
|
|
299
|
+
int descriptor = open(path, O_RDONLY);
|
|
300
|
+
if (descriptor >= 0) {
|
|
301
|
+
(void)fsync(descriptor);
|
|
302
|
+
(void)close(descriptor);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
static bool parse_segment_name(const char *name, uint64_t *out_id)
|
|
307
|
+
{
|
|
308
|
+
static const char prefix[] = "segment-";
|
|
309
|
+
static const char suffix[] = ".sfs";
|
|
310
|
+
const size_t expected_length = (sizeof(prefix) - 1u) + 20u +
|
|
311
|
+
(sizeof(suffix) - 1u);
|
|
312
|
+
uint64_t value = 0u;
|
|
313
|
+
size_t index;
|
|
314
|
+
if (strlen(name) != expected_length ||
|
|
315
|
+
memcmp(name, prefix, sizeof(prefix) - 1u) != 0 ||
|
|
316
|
+
memcmp(name + expected_length - (sizeof(suffix) - 1u),
|
|
317
|
+
suffix,
|
|
318
|
+
sizeof(suffix) - 1u) != 0) {
|
|
319
|
+
return false;
|
|
320
|
+
}
|
|
321
|
+
for (index = sizeof(prefix) - 1u;
|
|
322
|
+
index < (sizeof(prefix) - 1u) + 20u;
|
|
323
|
+
++index) {
|
|
324
|
+
uint8_t digit;
|
|
325
|
+
if (name[index] < '0' || name[index] > '9') {
|
|
326
|
+
return false;
|
|
327
|
+
}
|
|
328
|
+
digit = (uint8_t)(name[index] - '0');
|
|
329
|
+
if (value > (UINT64_MAX - digit) / 10u) {
|
|
330
|
+
return false;
|
|
331
|
+
}
|
|
332
|
+
value = value * 10u + digit;
|
|
333
|
+
}
|
|
334
|
+
if (value == 0u) {
|
|
335
|
+
return false;
|
|
336
|
+
}
|
|
337
|
+
*out_id = value;
|
|
338
|
+
return true;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
static bool is_temporary_segment_name(const char *name)
|
|
342
|
+
{
|
|
343
|
+
static const char prefix[] = ".segment-";
|
|
344
|
+
static const char suffix[] = ".creating";
|
|
345
|
+
const size_t expected_length = (sizeof(prefix) - 1u) + 20u +
|
|
346
|
+
(sizeof(suffix) - 1u);
|
|
347
|
+
size_t index;
|
|
348
|
+
if (strlen(name) != expected_length ||
|
|
349
|
+
memcmp(name, prefix, sizeof(prefix) - 1u) != 0 ||
|
|
350
|
+
memcmp(name + expected_length - (sizeof(suffix) - 1u),
|
|
351
|
+
suffix,
|
|
352
|
+
sizeof(suffix) - 1u) != 0) {
|
|
353
|
+
return false;
|
|
354
|
+
}
|
|
355
|
+
for (index = sizeof(prefix) - 1u;
|
|
356
|
+
index < (sizeof(prefix) - 1u) + 20u;
|
|
357
|
+
++index) {
|
|
358
|
+
if (name[index] < '0' || name[index] > '9') {
|
|
359
|
+
return false;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
return true;
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
static sfs_result_t cleanup_temporary_segments(sfs_partition_t *partition,
|
|
366
|
+
sfs_error_t *error)
|
|
367
|
+
{
|
|
368
|
+
DIR *directory = opendir(partition->directory);
|
|
369
|
+
struct dirent *entry;
|
|
370
|
+
bool removed = false;
|
|
371
|
+
if (directory == NULL) {
|
|
372
|
+
int saved_errno = errno;
|
|
373
|
+
return set_error(error,
|
|
374
|
+
SFS_ERR_IO,
|
|
375
|
+
saved_errno,
|
|
376
|
+
"cannot inspect temporary segments: %s",
|
|
377
|
+
strerror(saved_errno));
|
|
378
|
+
}
|
|
379
|
+
while ((entry = readdir(directory)) != NULL) {
|
|
380
|
+
char *path;
|
|
381
|
+
if (!is_temporary_segment_name(entry->d_name)) {
|
|
382
|
+
continue;
|
|
383
|
+
}
|
|
384
|
+
path = path_join(partition->directory, entry->d_name);
|
|
385
|
+
if (path == NULL) {
|
|
386
|
+
(void)closedir(directory);
|
|
387
|
+
return set_error(error,
|
|
388
|
+
SFS_ERR_NOMEM,
|
|
389
|
+
errno,
|
|
390
|
+
"cannot allocate temporary segment path");
|
|
391
|
+
}
|
|
392
|
+
if (unlink(path) != 0 && errno != ENOENT) {
|
|
393
|
+
int saved_errno = errno;
|
|
394
|
+
free(path);
|
|
395
|
+
(void)closedir(directory);
|
|
396
|
+
return set_error(error,
|
|
397
|
+
SFS_ERR_IO,
|
|
398
|
+
saved_errno,
|
|
399
|
+
"cannot remove interrupted segment creation: %s",
|
|
400
|
+
strerror(saved_errno));
|
|
401
|
+
}
|
|
402
|
+
free(path);
|
|
403
|
+
removed = true;
|
|
404
|
+
}
|
|
405
|
+
(void)closedir(directory);
|
|
406
|
+
if (removed) {
|
|
407
|
+
fsync_directory_best_effort(partition->directory);
|
|
408
|
+
}
|
|
409
|
+
return SFS_OK;
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
static int compare_u64(const void *left, const void *right)
|
|
413
|
+
{
|
|
414
|
+
uint64_t left_value = *(const uint64_t *)left;
|
|
415
|
+
uint64_t right_value = *(const uint64_t *)right;
|
|
416
|
+
return left_value < right_value ? -1 : (left_value > right_value ? 1 : 0);
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
static sfs_result_t reserve_segments(sfs_partition_t *partition,
|
|
420
|
+
size_t required,
|
|
421
|
+
sfs_error_t *error)
|
|
422
|
+
{
|
|
423
|
+
sfs_segment_meta_t *resized;
|
|
424
|
+
size_t capacity;
|
|
425
|
+
if (required <= partition->segment_capacity) {
|
|
426
|
+
return SFS_OK;
|
|
427
|
+
}
|
|
428
|
+
capacity = partition->segment_capacity == 0u ? 4u : partition->segment_capacity;
|
|
429
|
+
while (capacity < required) {
|
|
430
|
+
if (capacity > SIZE_MAX / 2u) {
|
|
431
|
+
return set_error(error, SFS_ERR_NOMEM, 0, "segment index is too large");
|
|
432
|
+
}
|
|
433
|
+
capacity *= 2u;
|
|
434
|
+
}
|
|
435
|
+
resized = (sfs_segment_meta_t *)realloc(
|
|
436
|
+
partition->segments, capacity * sizeof(*partition->segments));
|
|
437
|
+
if (resized == NULL) {
|
|
438
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot grow segment index");
|
|
439
|
+
}
|
|
440
|
+
partition->segments = resized;
|
|
441
|
+
partition->segment_capacity = capacity;
|
|
442
|
+
return SFS_OK;
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
static bool manifest_slot_decode(const uint8_t *slot,
|
|
446
|
+
sfs_loaded_manifest_t *manifest)
|
|
447
|
+
{
|
|
448
|
+
uint32_t partition_id;
|
|
449
|
+
if (memcmp(slot, SFS_MANIFEST_MAGIC, sizeof(SFS_MANIFEST_MAGIC)) != 0 ||
|
|
450
|
+
load_u32_le(slot + 8u) != SFS_FORMAT_VERSION ||
|
|
451
|
+
load_u32_le(slot + 12u) != SFS_MANIFEST_SLOT_SIZE ||
|
|
452
|
+
load_u32_le(slot + 40u) != SFS_MAX_PARTITIONS ||
|
|
453
|
+
load_u64_le(slot + SFS_MANIFEST_COMMIT_OFFSET) !=
|
|
454
|
+
SFS_MANIFEST_COMMIT_MARKER ||
|
|
455
|
+
load_u32_le(slot + SFS_MANIFEST_CRC_OFFSET) !=
|
|
456
|
+
sfs_crc32c(slot, SFS_MANIFEST_CRC_OFFSET)) {
|
|
457
|
+
return false;
|
|
458
|
+
}
|
|
459
|
+
memset(manifest, 0, sizeof(*manifest));
|
|
460
|
+
manifest->generation = load_u64_le(slot + 16u);
|
|
461
|
+
manifest->next_sequence = load_u64_le(slot + 24u);
|
|
462
|
+
manifest->segment_size = load_u64_le(slot + 32u);
|
|
463
|
+
if (manifest->generation == 0u || manifest->next_sequence == 0u) {
|
|
464
|
+
return false;
|
|
465
|
+
}
|
|
466
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
467
|
+
const uint8_t *entry = slot + SFS_MANIFEST_PARTITIONS_OFFSET +
|
|
468
|
+
partition_id * SFS_MANIFEST_PARTITION_SIZE;
|
|
469
|
+
sfs_loaded_manifest_partition_t *target =
|
|
470
|
+
&manifest->partitions[partition_id];
|
|
471
|
+
target->quota_bytes = load_u64_le(entry);
|
|
472
|
+
target->next_segment_id = load_u64_le(entry + 8u);
|
|
473
|
+
target->first_retained_segment_id = load_u64_le(entry + 16u);
|
|
474
|
+
target->evicted_segments = load_u64_le(entry + 24u);
|
|
475
|
+
target->evicted_records = load_u64_le(entry + 32u);
|
|
476
|
+
target->evicted_payload_bytes = load_u64_le(entry + 40u);
|
|
477
|
+
if (target->quota_bytes != 0u &&
|
|
478
|
+
(target->next_segment_id == 0u ||
|
|
479
|
+
target->first_retained_segment_id == 0u)) {
|
|
480
|
+
return false;
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
return true;
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
static sfs_result_t manifest_save(sfs_store_t *store, sfs_error_t *error)
|
|
487
|
+
{
|
|
488
|
+
int target_slot = store->manifest_active_slot < 0
|
|
489
|
+
? 0
|
|
490
|
+
: (store->manifest_active_slot + 1) %
|
|
491
|
+
(int)SFS_MANIFEST_SLOT_COUNT;
|
|
492
|
+
uint8_t *slot = store->manifest_map +
|
|
493
|
+
(size_t)target_slot * SFS_MANIFEST_SLOT_SIZE;
|
|
494
|
+
uint32_t partition_id;
|
|
495
|
+
uint64_t next_generation = store->manifest_generation + 1u;
|
|
496
|
+
if (next_generation == 0u) {
|
|
497
|
+
return set_error(error,
|
|
498
|
+
SFS_ERR_FULL,
|
|
499
|
+
0,
|
|
500
|
+
"manifest generation is exhausted");
|
|
501
|
+
}
|
|
502
|
+
memset(slot, 0, SFS_MANIFEST_SLOT_SIZE);
|
|
503
|
+
memcpy(slot, SFS_MANIFEST_MAGIC, sizeof(SFS_MANIFEST_MAGIC));
|
|
504
|
+
store_u32_le(slot + 8u, SFS_FORMAT_VERSION);
|
|
505
|
+
store_u32_le(slot + 12u, SFS_MANIFEST_SLOT_SIZE);
|
|
506
|
+
store_u64_le(slot + 16u, next_generation);
|
|
507
|
+
store_u64_le(slot + 24u, store->next_sequence);
|
|
508
|
+
store_u64_le(slot + 32u, store->segment_size);
|
|
509
|
+
store_u32_le(slot + 40u, SFS_MAX_PARTITIONS);
|
|
510
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
511
|
+
uint8_t *entry = slot + SFS_MANIFEST_PARTITIONS_OFFSET +
|
|
512
|
+
partition_id * SFS_MANIFEST_PARTITION_SIZE;
|
|
513
|
+
const sfs_partition_t *source = &store->partitions[partition_id];
|
|
514
|
+
store_u64_le(entry, source->quota_bytes);
|
|
515
|
+
store_u64_le(entry + 8u, source->next_segment_id);
|
|
516
|
+
store_u64_le(entry + 16u, source->first_retained_segment_id);
|
|
517
|
+
store_u64_le(entry + 24u, source->evicted_segments);
|
|
518
|
+
store_u64_le(entry + 32u, source->evicted_records);
|
|
519
|
+
store_u64_le(entry + 40u, source->evicted_payload_bytes);
|
|
520
|
+
}
|
|
521
|
+
store_u32_le(slot + SFS_MANIFEST_CRC_OFFSET,
|
|
522
|
+
sfs_crc32c(slot, SFS_MANIFEST_CRC_OFFSET));
|
|
523
|
+
atomic_thread_fence(memory_order_release);
|
|
524
|
+
if (msync(store->manifest_map, SFS_MANIFEST_SIZE, MS_SYNC) != 0) {
|
|
525
|
+
int saved_errno = errno;
|
|
526
|
+
return set_error(error,
|
|
527
|
+
SFS_ERR_IO,
|
|
528
|
+
saved_errno,
|
|
529
|
+
"cannot flush manifest body: %s",
|
|
530
|
+
strerror(saved_errno));
|
|
531
|
+
}
|
|
532
|
+
store_u64_le(slot + SFS_MANIFEST_COMMIT_OFFSET,
|
|
533
|
+
SFS_MANIFEST_COMMIT_MARKER);
|
|
534
|
+
atomic_thread_fence(memory_order_release);
|
|
535
|
+
if (msync(store->manifest_map, SFS_MANIFEST_SIZE, MS_SYNC) != 0 ||
|
|
536
|
+
fsync(store->manifest_fd) != 0) {
|
|
537
|
+
int saved_errno = errno;
|
|
538
|
+
return set_error(error,
|
|
539
|
+
SFS_ERR_IO,
|
|
540
|
+
saved_errno,
|
|
541
|
+
"cannot commit manifest: %s",
|
|
542
|
+
strerror(saved_errno));
|
|
543
|
+
}
|
|
544
|
+
store->manifest_active_slot = target_slot;
|
|
545
|
+
store->manifest_generation = next_generation;
|
|
546
|
+
return SFS_OK;
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
static sfs_result_t lock_store_directory(sfs_store_t *store, sfs_error_t *error)
|
|
550
|
+
{
|
|
551
|
+
char *path = path_join(store->directory, SFS_LOCK_FILE);
|
|
552
|
+
if (path == NULL) {
|
|
553
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot allocate lock path");
|
|
554
|
+
}
|
|
555
|
+
store->lock_fd = open(path, O_RDWR | O_CREAT, 0600);
|
|
556
|
+
free(path);
|
|
557
|
+
if (store->lock_fd < 0) {
|
|
558
|
+
int saved_errno = errno;
|
|
559
|
+
return set_error(error,
|
|
560
|
+
SFS_ERR_IO,
|
|
561
|
+
saved_errno,
|
|
562
|
+
"cannot open store lock: %s",
|
|
563
|
+
strerror(saved_errno));
|
|
564
|
+
}
|
|
565
|
+
if (flock(store->lock_fd, LOCK_EX | LOCK_NB) != 0) {
|
|
566
|
+
int saved_errno = errno;
|
|
567
|
+
return set_error(error,
|
|
568
|
+
(saved_errno == EWOULDBLOCK || saved_errno == EAGAIN ||
|
|
569
|
+
saved_errno == EACCES)
|
|
570
|
+
? SFS_ERR_BUSY
|
|
571
|
+
: SFS_ERR_IO,
|
|
572
|
+
saved_errno,
|
|
573
|
+
"store writer lock is busy");
|
|
574
|
+
}
|
|
575
|
+
return SFS_OK;
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
static sfs_result_t open_manifest(sfs_store_t *store,
|
|
579
|
+
const sfs_open_options_t *options,
|
|
580
|
+
sfs_error_t *error)
|
|
581
|
+
{
|
|
582
|
+
char *path = path_join(store->directory, SFS_MANIFEST_FILE);
|
|
583
|
+
struct stat status;
|
|
584
|
+
bool new_manifest;
|
|
585
|
+
int open_flags = O_RDWR;
|
|
586
|
+
sfs_loaded_manifest_t decoded[2];
|
|
587
|
+
bool valid[2] = {false, false};
|
|
588
|
+
int selected = -1;
|
|
589
|
+
uint32_t partition_id;
|
|
590
|
+
if (path == NULL) {
|
|
591
|
+
return set_error(error,
|
|
592
|
+
SFS_ERR_NOMEM,
|
|
593
|
+
errno,
|
|
594
|
+
"cannot allocate manifest path");
|
|
595
|
+
}
|
|
596
|
+
if ((options->flags & SFS_OPEN_CREATE) != 0u) {
|
|
597
|
+
open_flags |= O_CREAT;
|
|
598
|
+
}
|
|
599
|
+
store->manifest_fd = open(path, open_flags, 0600);
|
|
600
|
+
free(path);
|
|
601
|
+
if (store->manifest_fd < 0) {
|
|
602
|
+
int saved_errno = errno;
|
|
603
|
+
return set_error(error,
|
|
604
|
+
SFS_ERR_IO,
|
|
605
|
+
saved_errno,
|
|
606
|
+
"cannot open manifest: %s",
|
|
607
|
+
strerror(saved_errno));
|
|
608
|
+
}
|
|
609
|
+
if (fstat(store->manifest_fd, &status) != 0) {
|
|
610
|
+
int saved_errno = errno;
|
|
611
|
+
return set_error(error,
|
|
612
|
+
SFS_ERR_IO,
|
|
613
|
+
saved_errno,
|
|
614
|
+
"cannot stat manifest: %s",
|
|
615
|
+
strerror(saved_errno));
|
|
616
|
+
}
|
|
617
|
+
new_manifest = status.st_size == 0;
|
|
618
|
+
if (!new_manifest && status.st_size != (off_t)SFS_MANIFEST_SIZE) {
|
|
619
|
+
return set_error(error,
|
|
620
|
+
SFS_ERR_CORRUPT,
|
|
621
|
+
0,
|
|
622
|
+
"manifest size is invalid");
|
|
623
|
+
}
|
|
624
|
+
if (new_manifest) {
|
|
625
|
+
bool any_partition = false;
|
|
626
|
+
if ((options->flags & SFS_OPEN_CREATE) == 0u) {
|
|
627
|
+
return set_error(error,
|
|
628
|
+
SFS_ERR_IO,
|
|
629
|
+
ENOENT,
|
|
630
|
+
"store does not exist");
|
|
631
|
+
}
|
|
632
|
+
store->segment_size = options->segment_size == 0u
|
|
633
|
+
? SFS_DEFAULT_SEGMENT_SIZE
|
|
634
|
+
: options->segment_size;
|
|
635
|
+
if (store->segment_size < SFS_MIN_SEGMENT_SIZE ||
|
|
636
|
+
store->segment_size > SIZE_MAX) {
|
|
637
|
+
return set_error(error,
|
|
638
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
639
|
+
0,
|
|
640
|
+
"segment_size must be between %u and SIZE_MAX",
|
|
641
|
+
SFS_MIN_SEGMENT_SIZE);
|
|
642
|
+
}
|
|
643
|
+
store->next_sequence = 1u;
|
|
644
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS;
|
|
645
|
+
++partition_id) {
|
|
646
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
647
|
+
partition->quota_bytes = options->partition_quotas[partition_id];
|
|
648
|
+
partition->enabled = partition->quota_bytes != 0u;
|
|
649
|
+
partition->next_segment_id = 1u;
|
|
650
|
+
partition->first_retained_segment_id = 1u;
|
|
651
|
+
if (partition->enabled) {
|
|
652
|
+
any_partition = true;
|
|
653
|
+
if (partition->quota_bytes < store->segment_size) {
|
|
654
|
+
return set_error(error,
|
|
655
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
656
|
+
0,
|
|
657
|
+
"partition %u quota is smaller than one segment",
|
|
658
|
+
partition_id);
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
if (!any_partition) {
|
|
663
|
+
return set_error(error,
|
|
664
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
665
|
+
0,
|
|
666
|
+
"at least one partition quota must be non-zero");
|
|
667
|
+
}
|
|
668
|
+
if (ftruncate(store->manifest_fd, (off_t)SFS_MANIFEST_SIZE) != 0) {
|
|
669
|
+
int saved_errno = errno;
|
|
670
|
+
return set_error(error,
|
|
671
|
+
SFS_ERR_IO,
|
|
672
|
+
saved_errno,
|
|
673
|
+
"cannot size manifest: %s",
|
|
674
|
+
strerror(saved_errno));
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
store->manifest_map = (uint8_t *)mmap(NULL,
|
|
678
|
+
SFS_MANIFEST_SIZE,
|
|
679
|
+
PROT_READ | PROT_WRITE,
|
|
680
|
+
MAP_SHARED,
|
|
681
|
+
store->manifest_fd,
|
|
682
|
+
0);
|
|
683
|
+
if (store->manifest_map == MAP_FAILED) {
|
|
684
|
+
int saved_errno = errno;
|
|
685
|
+
store->manifest_map = NULL;
|
|
686
|
+
return set_error(error,
|
|
687
|
+
SFS_ERR_IO,
|
|
688
|
+
saved_errno,
|
|
689
|
+
"cannot map manifest: %s",
|
|
690
|
+
strerror(saved_errno));
|
|
691
|
+
}
|
|
692
|
+
if (new_manifest) {
|
|
693
|
+
memset(store->manifest_map, 0, SFS_MANIFEST_SIZE);
|
|
694
|
+
store->manifest_active_slot = -1;
|
|
695
|
+
store->manifest_generation = 0u;
|
|
696
|
+
return manifest_save(store, error);
|
|
697
|
+
}
|
|
698
|
+
valid[0] = manifest_slot_decode(store->manifest_map, &decoded[0]);
|
|
699
|
+
valid[1] = manifest_slot_decode(store->manifest_map + SFS_MANIFEST_SLOT_SIZE,
|
|
700
|
+
&decoded[1]);
|
|
701
|
+
if (valid[0] && valid[1]) {
|
|
702
|
+
selected = decoded[1].generation > decoded[0].generation ? 1 : 0;
|
|
703
|
+
} else if (valid[0]) {
|
|
704
|
+
selected = 0;
|
|
705
|
+
} else if (valid[1]) {
|
|
706
|
+
selected = 1;
|
|
707
|
+
} else {
|
|
708
|
+
uint32_t version0 = load_u32_le(store->manifest_map + 8u);
|
|
709
|
+
uint32_t version1 = load_u32_le(store->manifest_map +
|
|
710
|
+
SFS_MANIFEST_SLOT_SIZE + 8u);
|
|
711
|
+
return set_error(error,
|
|
712
|
+
(version0 != 0u && version0 != SFS_FORMAT_VERSION) ||
|
|
713
|
+
(version1 != 0u && version1 != SFS_FORMAT_VERSION)
|
|
714
|
+
? SFS_ERR_FORMAT_VERSION
|
|
715
|
+
: SFS_ERR_CORRUPT,
|
|
716
|
+
0,
|
|
717
|
+
"manifest has no valid committed slot");
|
|
718
|
+
}
|
|
719
|
+
store->manifest_active_slot = selected;
|
|
720
|
+
store->manifest_generation = decoded[selected].generation;
|
|
721
|
+
store->segment_size = decoded[selected].segment_size;
|
|
722
|
+
store->next_sequence = decoded[selected].next_sequence;
|
|
723
|
+
if ((options->segment_size != 0u &&
|
|
724
|
+
options->segment_size != store->segment_size) ||
|
|
725
|
+
store->segment_size < SFS_MIN_SEGMENT_SIZE ||
|
|
726
|
+
store->segment_size > SIZE_MAX) {
|
|
727
|
+
return set_error(error,
|
|
728
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
729
|
+
0,
|
|
730
|
+
"configured segment_size does not match the store");
|
|
731
|
+
}
|
|
732
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
733
|
+
const sfs_loaded_manifest_partition_t *source =
|
|
734
|
+
&decoded[selected].partitions[partition_id];
|
|
735
|
+
sfs_partition_t *target = &store->partitions[partition_id];
|
|
736
|
+
if (options->partition_quotas[partition_id] != 0u &&
|
|
737
|
+
options->partition_quotas[partition_id] != source->quota_bytes) {
|
|
738
|
+
return set_error(error,
|
|
739
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
740
|
+
0,
|
|
741
|
+
"partition %u quota does not match the store",
|
|
742
|
+
partition_id);
|
|
743
|
+
}
|
|
744
|
+
target->quota_bytes = source->quota_bytes;
|
|
745
|
+
target->enabled = source->quota_bytes != 0u;
|
|
746
|
+
target->next_segment_id = source->next_segment_id;
|
|
747
|
+
target->first_retained_segment_id = source->first_retained_segment_id;
|
|
748
|
+
target->evicted_segments = source->evicted_segments;
|
|
749
|
+
target->evicted_records = source->evicted_records;
|
|
750
|
+
target->evicted_payload_bytes = source->evicted_payload_bytes;
|
|
751
|
+
}
|
|
752
|
+
return SFS_OK;
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
static void encode_segment_header(uint8_t *header,
|
|
756
|
+
uint32_t partition_id,
|
|
757
|
+
uint64_t segment_id,
|
|
758
|
+
uint64_t segment_size,
|
|
759
|
+
uint64_t first_sequence)
|
|
760
|
+
{
|
|
761
|
+
memset(header, 0, SFS_SEGMENT_HEADER_SIZE);
|
|
762
|
+
memcpy(header, SFS_SEGMENT_MAGIC, sizeof(SFS_SEGMENT_MAGIC));
|
|
763
|
+
store_u32_le(header + 8u, SFS_FORMAT_VERSION);
|
|
764
|
+
store_u32_le(header + 12u, SFS_SEGMENT_HEADER_SIZE);
|
|
765
|
+
store_u64_le(header + 16u, segment_id);
|
|
766
|
+
store_u64_le(header + 24u, segment_size);
|
|
767
|
+
store_u32_le(header + 32u, partition_id);
|
|
768
|
+
store_u64_le(header + 40u, first_sequence);
|
|
769
|
+
store_u32_le(header + SFS_SEGMENT_CRC_OFFSET,
|
|
770
|
+
sfs_crc32c(header, SFS_SEGMENT_CRC_OFFSET));
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
static sfs_result_t validate_segment_header(const uint8_t *header,
|
|
774
|
+
uint32_t partition_id,
|
|
775
|
+
uint64_t segment_id,
|
|
776
|
+
uint64_t segment_size,
|
|
777
|
+
uint64_t *out_first_sequence,
|
|
778
|
+
sfs_error_t *error)
|
|
779
|
+
{
|
|
780
|
+
uint32_t version;
|
|
781
|
+
if (memcmp(header, SFS_SEGMENT_MAGIC, sizeof(SFS_SEGMENT_MAGIC)) != 0) {
|
|
782
|
+
return set_error(error,
|
|
783
|
+
SFS_ERR_CORRUPT,
|
|
784
|
+
0,
|
|
785
|
+
"segment %" PRIu64 " has invalid magic",
|
|
786
|
+
segment_id);
|
|
787
|
+
}
|
|
788
|
+
version = load_u32_le(header + 8u);
|
|
789
|
+
if (version != SFS_FORMAT_VERSION) {
|
|
790
|
+
return set_error(error,
|
|
791
|
+
SFS_ERR_FORMAT_VERSION,
|
|
792
|
+
0,
|
|
793
|
+
"segment %" PRIu64 " uses format version %u",
|
|
794
|
+
segment_id,
|
|
795
|
+
version);
|
|
796
|
+
}
|
|
797
|
+
if (load_u32_le(header + 12u) != SFS_SEGMENT_HEADER_SIZE ||
|
|
798
|
+
load_u64_le(header + 16u) != segment_id ||
|
|
799
|
+
load_u64_le(header + 24u) != segment_size ||
|
|
800
|
+
load_u32_le(header + 32u) != partition_id ||
|
|
801
|
+
load_u32_le(header + SFS_SEGMENT_CRC_OFFSET) !=
|
|
802
|
+
sfs_crc32c(header, SFS_SEGMENT_CRC_OFFSET)) {
|
|
803
|
+
return set_error(error,
|
|
804
|
+
SFS_ERR_CORRUPT,
|
|
805
|
+
0,
|
|
806
|
+
"segment %" PRIu64 " header is corrupt",
|
|
807
|
+
segment_id);
|
|
808
|
+
}
|
|
809
|
+
*out_first_sequence = load_u64_le(header + 40u);
|
|
810
|
+
if (*out_first_sequence == 0u) {
|
|
811
|
+
return set_error(error,
|
|
812
|
+
SFS_ERR_CORRUPT,
|
|
813
|
+
0,
|
|
814
|
+
"segment %" PRIu64 " has invalid first sequence",
|
|
815
|
+
segment_id);
|
|
816
|
+
}
|
|
817
|
+
return SFS_OK;
|
|
818
|
+
}
|
|
819
|
+
|
|
820
|
+
static bool committed_marker_follows(const uint8_t *mapping,
|
|
821
|
+
uint64_t file_size,
|
|
822
|
+
uint64_t segment_size,
|
|
823
|
+
uint64_t frame_offset)
|
|
824
|
+
{
|
|
825
|
+
uint64_t limit = file_size < segment_size ? file_size : segment_size;
|
|
826
|
+
uint64_t marker_offset = frame_offset + SFS_FRAME_PREFIX_SIZE;
|
|
827
|
+
while (marker_offset <= limit && limit - marker_offset >= SFS_FRAME_COMMIT_SIZE) {
|
|
828
|
+
if (load_u64_le(mapping + marker_offset) == SFS_FRAME_COMMIT_MARKER) {
|
|
829
|
+
return true;
|
|
830
|
+
}
|
|
831
|
+
marker_offset += 8u;
|
|
832
|
+
}
|
|
833
|
+
return false;
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
static int decode_frame(const uint8_t *mapping,
|
|
837
|
+
uint64_t file_size,
|
|
838
|
+
uint64_t segment_size,
|
|
839
|
+
uint64_t offset,
|
|
840
|
+
sfs_frame_view_t *frame,
|
|
841
|
+
sfs_error_t *error)
|
|
842
|
+
{
|
|
843
|
+
const uint8_t *prefix;
|
|
844
|
+
uint64_t expected_length;
|
|
845
|
+
uint64_t index;
|
|
846
|
+
bool prefix_zero = true;
|
|
847
|
+
if (offset == file_size) {
|
|
848
|
+
return SFS_END;
|
|
849
|
+
}
|
|
850
|
+
if (offset > file_size) {
|
|
851
|
+
return SFS_INTERNAL_TORN;
|
|
852
|
+
}
|
|
853
|
+
if (file_size - offset < SFS_FRAME_PREFIX_SIZE) {
|
|
854
|
+
/*
|
|
855
|
+
* A valid frame can leave 1..23 bytes at the end of a fixed-size
|
|
856
|
+
* segment. A clean preallocated tail is all zero and is a normal EOF,
|
|
857
|
+
* including after the segment has been sealed by rotation. Preserve
|
|
858
|
+
* strict torn-write detection when any byte of that short tail was
|
|
859
|
+
* actually touched.
|
|
860
|
+
*/
|
|
861
|
+
for (index = offset; index < file_size; ++index) {
|
|
862
|
+
if (mapping[index] != 0u) {
|
|
863
|
+
return SFS_INTERNAL_TORN;
|
|
864
|
+
}
|
|
865
|
+
}
|
|
866
|
+
return SFS_END;
|
|
867
|
+
}
|
|
868
|
+
prefix = mapping + offset;
|
|
869
|
+
for (index = 0u; index < SFS_FRAME_PREFIX_SIZE; ++index) {
|
|
870
|
+
if (prefix[index] != 0u) {
|
|
871
|
+
prefix_zero = false;
|
|
872
|
+
break;
|
|
873
|
+
}
|
|
874
|
+
}
|
|
875
|
+
if (prefix_zero) {
|
|
876
|
+
return committed_marker_follows(mapping,
|
|
877
|
+
file_size,
|
|
878
|
+
segment_size,
|
|
879
|
+
offset)
|
|
880
|
+
? set_error(error,
|
|
881
|
+
SFS_ERR_CORRUPT,
|
|
882
|
+
0,
|
|
883
|
+
"committed frame header at offset %" PRIu64
|
|
884
|
+
" was cleared",
|
|
885
|
+
offset)
|
|
886
|
+
: SFS_END;
|
|
887
|
+
}
|
|
888
|
+
frame->total_length = load_u32_le(prefix);
|
|
889
|
+
frame->payload_length = load_u32_le(prefix + 4u);
|
|
890
|
+
frame->sequence = load_u64_le(prefix + 8u);
|
|
891
|
+
if (load_u32_le(prefix + 20u) != sfs_crc32c(prefix, 20u)) {
|
|
892
|
+
return committed_marker_follows(mapping,
|
|
893
|
+
file_size,
|
|
894
|
+
segment_size,
|
|
895
|
+
offset)
|
|
896
|
+
? set_error(error,
|
|
897
|
+
SFS_ERR_CORRUPT,
|
|
898
|
+
0,
|
|
899
|
+
"committed frame header at offset %" PRIu64
|
|
900
|
+
" failed CRC32C",
|
|
901
|
+
offset)
|
|
902
|
+
: SFS_INTERNAL_TORN;
|
|
903
|
+
}
|
|
904
|
+
expected_length = align_eight(SFS_FRAME_PREFIX_SIZE +
|
|
905
|
+
(uint64_t)frame->payload_length) +
|
|
906
|
+
SFS_FRAME_COMMIT_SIZE;
|
|
907
|
+
if (frame->total_length != expected_length ||
|
|
908
|
+
frame->total_length < SFS_MIN_FRAME_SIZE ||
|
|
909
|
+
(frame->total_length & 7u) != 0u ||
|
|
910
|
+
frame->total_length > segment_size - offset) {
|
|
911
|
+
return committed_marker_follows(mapping,
|
|
912
|
+
file_size,
|
|
913
|
+
segment_size,
|
|
914
|
+
offset)
|
|
915
|
+
? set_error(error,
|
|
916
|
+
SFS_ERR_CORRUPT,
|
|
917
|
+
0,
|
|
918
|
+
"committed frame length at offset %" PRIu64
|
|
919
|
+
" is corrupt",
|
|
920
|
+
offset)
|
|
921
|
+
: SFS_INTERNAL_TORN;
|
|
922
|
+
}
|
|
923
|
+
frame->payload_offset = offset + SFS_FRAME_PREFIX_SIZE;
|
|
924
|
+
frame->commit_offset = offset + frame->total_length - SFS_FRAME_COMMIT_SIZE;
|
|
925
|
+
if (offset + frame->total_length > file_size) {
|
|
926
|
+
return SFS_INTERNAL_TORN;
|
|
927
|
+
}
|
|
928
|
+
if (load_u64_le(mapping + frame->commit_offset) !=
|
|
929
|
+
SFS_FRAME_COMMIT_MARKER) {
|
|
930
|
+
return SFS_INTERNAL_TORN;
|
|
931
|
+
}
|
|
932
|
+
if (frame->sequence == 0u ||
|
|
933
|
+
load_u32_le(prefix + 16u) !=
|
|
934
|
+
sfs_crc32c(mapping + frame->payload_offset,
|
|
935
|
+
frame->payload_length)) {
|
|
936
|
+
return set_error(error,
|
|
937
|
+
SFS_ERR_CORRUPT,
|
|
938
|
+
0,
|
|
939
|
+
"committed frame at offset %" PRIu64 " failed CRC32C",
|
|
940
|
+
offset);
|
|
941
|
+
}
|
|
942
|
+
for (index = frame->payload_offset + frame->payload_length;
|
|
943
|
+
index < frame->commit_offset;
|
|
944
|
+
++index) {
|
|
945
|
+
if (mapping[index] != 0u) {
|
|
946
|
+
return set_error(error,
|
|
947
|
+
SFS_ERR_CORRUPT,
|
|
948
|
+
0,
|
|
949
|
+
"committed frame at offset %" PRIu64
|
|
950
|
+
" has non-zero alignment padding",
|
|
951
|
+
offset);
|
|
952
|
+
}
|
|
953
|
+
}
|
|
954
|
+
return SFS_OK;
|
|
955
|
+
}
|
|
956
|
+
|
|
957
|
+
static sfs_result_t scan_segment_file(sfs_store_t *store,
|
|
958
|
+
uint32_t partition_id,
|
|
959
|
+
uint64_t segment_id,
|
|
960
|
+
bool active,
|
|
961
|
+
sfs_segment_meta_t *meta,
|
|
962
|
+
sfs_error_t *error)
|
|
963
|
+
{
|
|
964
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
965
|
+
char *path = segment_path(partition, segment_id);
|
|
966
|
+
int descriptor;
|
|
967
|
+
struct stat status;
|
|
968
|
+
uint8_t *mapping;
|
|
969
|
+
uint64_t file_size;
|
|
970
|
+
uint64_t offset = SFS_SEGMENT_HEADER_SIZE;
|
|
971
|
+
uint64_t previous_sequence = 0u;
|
|
972
|
+
uint64_t header_first_sequence = 0u;
|
|
973
|
+
int frame_result;
|
|
974
|
+
if (path == NULL) {
|
|
975
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot allocate segment path");
|
|
976
|
+
}
|
|
977
|
+
descriptor = open(path, O_RDWR);
|
|
978
|
+
free(path);
|
|
979
|
+
if (descriptor < 0) {
|
|
980
|
+
int saved_errno = errno;
|
|
981
|
+
return set_error(error,
|
|
982
|
+
SFS_ERR_IO,
|
|
983
|
+
saved_errno,
|
|
984
|
+
"cannot open segment %" PRIu64 ": %s",
|
|
985
|
+
segment_id,
|
|
986
|
+
strerror(saved_errno));
|
|
987
|
+
}
|
|
988
|
+
if (fstat(descriptor, &status) != 0 || status.st_size < 0) {
|
|
989
|
+
int saved_errno = errno;
|
|
990
|
+
(void)close(descriptor);
|
|
991
|
+
return set_error(error,
|
|
992
|
+
SFS_ERR_IO,
|
|
993
|
+
saved_errno,
|
|
994
|
+
"cannot stat segment %" PRIu64,
|
|
995
|
+
segment_id);
|
|
996
|
+
}
|
|
997
|
+
file_size = (uint64_t)status.st_size;
|
|
998
|
+
if (file_size < SFS_SEGMENT_HEADER_SIZE || file_size > store->segment_size ||
|
|
999
|
+
(!active && file_size != store->segment_size)) {
|
|
1000
|
+
(void)close(descriptor);
|
|
1001
|
+
return set_error(error,
|
|
1002
|
+
SFS_ERR_CORRUPT,
|
|
1003
|
+
0,
|
|
1004
|
+
"segment %" PRIu64 " has invalid file size",
|
|
1005
|
+
segment_id);
|
|
1006
|
+
}
|
|
1007
|
+
mapping = (uint8_t *)mmap(NULL,
|
|
1008
|
+
(size_t)file_size,
|
|
1009
|
+
PROT_READ | PROT_WRITE,
|
|
1010
|
+
MAP_SHARED,
|
|
1011
|
+
descriptor,
|
|
1012
|
+
0);
|
|
1013
|
+
if (mapping == MAP_FAILED) {
|
|
1014
|
+
int saved_errno = errno;
|
|
1015
|
+
(void)close(descriptor);
|
|
1016
|
+
return set_error(error,
|
|
1017
|
+
SFS_ERR_IO,
|
|
1018
|
+
saved_errno,
|
|
1019
|
+
"cannot map segment %" PRIu64 ": %s",
|
|
1020
|
+
segment_id,
|
|
1021
|
+
strerror(saved_errno));
|
|
1022
|
+
}
|
|
1023
|
+
{
|
|
1024
|
+
sfs_result_t header_result = validate_segment_header(mapping,
|
|
1025
|
+
partition_id,
|
|
1026
|
+
segment_id,
|
|
1027
|
+
store->segment_size,
|
|
1028
|
+
&header_first_sequence,
|
|
1029
|
+
error);
|
|
1030
|
+
if (header_result != SFS_OK) {
|
|
1031
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1032
|
+
(void)close(descriptor);
|
|
1033
|
+
return header_result;
|
|
1034
|
+
}
|
|
1035
|
+
}
|
|
1036
|
+
memset(meta, 0, sizeof(*meta));
|
|
1037
|
+
meta->id = segment_id;
|
|
1038
|
+
for (;;) {
|
|
1039
|
+
sfs_frame_view_t frame;
|
|
1040
|
+
frame_result = decode_frame(mapping,
|
|
1041
|
+
file_size,
|
|
1042
|
+
store->segment_size,
|
|
1043
|
+
offset,
|
|
1044
|
+
&frame,
|
|
1045
|
+
error);
|
|
1046
|
+
if (frame_result == SFS_END || frame_result == SFS_INTERNAL_TORN) {
|
|
1047
|
+
break;
|
|
1048
|
+
}
|
|
1049
|
+
if (frame_result != SFS_OK) {
|
|
1050
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1051
|
+
(void)close(descriptor);
|
|
1052
|
+
return (sfs_result_t)frame_result;
|
|
1053
|
+
}
|
|
1054
|
+
if (previous_sequence != 0u && frame.sequence <= previous_sequence) {
|
|
1055
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1056
|
+
(void)close(descriptor);
|
|
1057
|
+
return set_error(error,
|
|
1058
|
+
SFS_ERR_CORRUPT,
|
|
1059
|
+
0,
|
|
1060
|
+
"segment %" PRIu64 " sequence order is corrupt",
|
|
1061
|
+
segment_id);
|
|
1062
|
+
}
|
|
1063
|
+
if (meta->record_count == 0u) {
|
|
1064
|
+
meta->first_sequence = frame.sequence;
|
|
1065
|
+
}
|
|
1066
|
+
meta->last_sequence = frame.sequence;
|
|
1067
|
+
meta->record_count += 1u;
|
|
1068
|
+
meta->payload_bytes += frame.payload_length;
|
|
1069
|
+
previous_sequence = frame.sequence;
|
|
1070
|
+
offset += frame.total_length;
|
|
1071
|
+
}
|
|
1072
|
+
meta->write_offset = offset;
|
|
1073
|
+
if (frame_result == SFS_INTERNAL_TORN && !active) {
|
|
1074
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1075
|
+
(void)close(descriptor);
|
|
1076
|
+
return set_error(error,
|
|
1077
|
+
SFS_ERR_CORRUPT,
|
|
1078
|
+
0,
|
|
1079
|
+
"sealed segment %" PRIu64 " has a torn tail",
|
|
1080
|
+
segment_id);
|
|
1081
|
+
}
|
|
1082
|
+
if (active && (frame_result == SFS_INTERNAL_TORN ||
|
|
1083
|
+
file_size != store->segment_size)) {
|
|
1084
|
+
uint64_t discarded = file_size > offset ? file_size - offset : 0u;
|
|
1085
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1086
|
+
if (ftruncate(descriptor, (off_t)store->segment_size) != 0) {
|
|
1087
|
+
int saved_errno = errno;
|
|
1088
|
+
(void)close(descriptor);
|
|
1089
|
+
return set_error(error,
|
|
1090
|
+
SFS_ERR_IO,
|
|
1091
|
+
saved_errno,
|
|
1092
|
+
"cannot restore segment %" PRIu64 " size: %s",
|
|
1093
|
+
segment_id,
|
|
1094
|
+
strerror(saved_errno));
|
|
1095
|
+
}
|
|
1096
|
+
mapping = (uint8_t *)mmap(NULL,
|
|
1097
|
+
(size_t)store->segment_size,
|
|
1098
|
+
PROT_READ | PROT_WRITE,
|
|
1099
|
+
MAP_SHARED,
|
|
1100
|
+
descriptor,
|
|
1101
|
+
0);
|
|
1102
|
+
if (mapping == MAP_FAILED) {
|
|
1103
|
+
int saved_errno = errno;
|
|
1104
|
+
(void)close(descriptor);
|
|
1105
|
+
return set_error(error,
|
|
1106
|
+
SFS_ERR_IO,
|
|
1107
|
+
saved_errno,
|
|
1108
|
+
"cannot remap recovered segment: %s",
|
|
1109
|
+
strerror(saved_errno));
|
|
1110
|
+
}
|
|
1111
|
+
memset(mapping + offset, 0, (size_t)(store->segment_size - offset));
|
|
1112
|
+
if (msync(mapping, (size_t)store->segment_size, MS_SYNC) != 0 ||
|
|
1113
|
+
fsync(descriptor) != 0) {
|
|
1114
|
+
int saved_errno = errno;
|
|
1115
|
+
(void)munmap(mapping, (size_t)store->segment_size);
|
|
1116
|
+
(void)close(descriptor);
|
|
1117
|
+
return set_error(error,
|
|
1118
|
+
SFS_ERR_IO,
|
|
1119
|
+
saved_errno,
|
|
1120
|
+
"cannot persist recovered tail: %s",
|
|
1121
|
+
strerror(saved_errno));
|
|
1122
|
+
}
|
|
1123
|
+
store->status_flags |= SFS_STATUS_RECOVERED_TAIL;
|
|
1124
|
+
store->recovery_partition_id = partition_id;
|
|
1125
|
+
store->recovery_segment_id = segment_id;
|
|
1126
|
+
store->recovery_offset = offset;
|
|
1127
|
+
store->recovery_discarded_bytes += discarded;
|
|
1128
|
+
file_size = store->segment_size;
|
|
1129
|
+
}
|
|
1130
|
+
if (active) {
|
|
1131
|
+
partition->active_fd = descriptor;
|
|
1132
|
+
partition->active_map = mapping;
|
|
1133
|
+
partition->active_write_offset = offset;
|
|
1134
|
+
partition->durable_offset = offset;
|
|
1135
|
+
} else {
|
|
1136
|
+
(void)munmap(mapping, (size_t)file_size);
|
|
1137
|
+
(void)close(descriptor);
|
|
1138
|
+
}
|
|
1139
|
+
return SFS_OK;
|
|
1140
|
+
}
|
|
1141
|
+
|
|
1142
|
+
static sfs_result_t enumerate_segment_ids(sfs_partition_t *partition,
|
|
1143
|
+
uint64_t **out_ids,
|
|
1144
|
+
size_t *out_count,
|
|
1145
|
+
sfs_error_t *error)
|
|
1146
|
+
{
|
|
1147
|
+
DIR *directory = opendir(partition->directory);
|
|
1148
|
+
struct dirent *entry;
|
|
1149
|
+
uint64_t *ids = NULL;
|
|
1150
|
+
size_t count = 0u;
|
|
1151
|
+
size_t capacity = 0u;
|
|
1152
|
+
if (directory == NULL) {
|
|
1153
|
+
int saved_errno = errno;
|
|
1154
|
+
return set_error(error,
|
|
1155
|
+
SFS_ERR_IO,
|
|
1156
|
+
saved_errno,
|
|
1157
|
+
"cannot list partition directory: %s",
|
|
1158
|
+
strerror(saved_errno));
|
|
1159
|
+
}
|
|
1160
|
+
while ((entry = readdir(directory)) != NULL) {
|
|
1161
|
+
uint64_t id;
|
|
1162
|
+
if (!parse_segment_name(entry->d_name, &id)) {
|
|
1163
|
+
continue;
|
|
1164
|
+
}
|
|
1165
|
+
if (count == capacity) {
|
|
1166
|
+
size_t new_capacity = capacity == 0u ? 8u : capacity * 2u;
|
|
1167
|
+
uint64_t *resized =
|
|
1168
|
+
(uint64_t *)realloc(ids, new_capacity * sizeof(*ids));
|
|
1169
|
+
if (resized == NULL) {
|
|
1170
|
+
free(ids);
|
|
1171
|
+
(void)closedir(directory);
|
|
1172
|
+
return set_error(error,
|
|
1173
|
+
SFS_ERR_NOMEM,
|
|
1174
|
+
errno,
|
|
1175
|
+
"cannot allocate segment id list");
|
|
1176
|
+
}
|
|
1177
|
+
ids = resized;
|
|
1178
|
+
capacity = new_capacity;
|
|
1179
|
+
}
|
|
1180
|
+
ids[count++] = id;
|
|
1181
|
+
}
|
|
1182
|
+
(void)closedir(directory);
|
|
1183
|
+
qsort(ids, count, sizeof(*ids), compare_u64);
|
|
1184
|
+
*out_ids = ids;
|
|
1185
|
+
*out_count = count;
|
|
1186
|
+
return SFS_OK;
|
|
1187
|
+
}
|
|
1188
|
+
|
|
1189
|
+
static sfs_result_t create_segment(sfs_store_t *store,
|
|
1190
|
+
uint32_t partition_id,
|
|
1191
|
+
sfs_error_t *error)
|
|
1192
|
+
{
|
|
1193
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1194
|
+
uint64_t segment_id = partition->next_segment_id;
|
|
1195
|
+
char *path;
|
|
1196
|
+
char *temporary_path;
|
|
1197
|
+
int descriptor;
|
|
1198
|
+
uint8_t *mapping;
|
|
1199
|
+
sfs_segment_meta_t *meta;
|
|
1200
|
+
sfs_result_t result;
|
|
1201
|
+
if (segment_id == 0u || segment_id == UINT64_MAX) {
|
|
1202
|
+
return set_error(error,
|
|
1203
|
+
SFS_ERR_FULL,
|
|
1204
|
+
0,
|
|
1205
|
+
"partition %u exhausted segment ids",
|
|
1206
|
+
partition_id);
|
|
1207
|
+
}
|
|
1208
|
+
result = reserve_segments(partition, partition->segment_count + 1u, error);
|
|
1209
|
+
if (result != SFS_OK) {
|
|
1210
|
+
return result;
|
|
1211
|
+
}
|
|
1212
|
+
path = segment_path(partition, segment_id);
|
|
1213
|
+
temporary_path = temporary_segment_path(partition, segment_id);
|
|
1214
|
+
if (path == NULL || temporary_path == NULL) {
|
|
1215
|
+
free(path);
|
|
1216
|
+
free(temporary_path);
|
|
1217
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot allocate segment path");
|
|
1218
|
+
}
|
|
1219
|
+
descriptor = open(temporary_path, O_RDWR | O_CREAT | O_EXCL, 0600);
|
|
1220
|
+
if (descriptor < 0) {
|
|
1221
|
+
int saved_errno = errno;
|
|
1222
|
+
free(path);
|
|
1223
|
+
free(temporary_path);
|
|
1224
|
+
return set_error(error,
|
|
1225
|
+
SFS_ERR_IO,
|
|
1226
|
+
saved_errno,
|
|
1227
|
+
"cannot create temporary segment %" PRIu64 ": %s",
|
|
1228
|
+
segment_id,
|
|
1229
|
+
strerror(saved_errno));
|
|
1230
|
+
}
|
|
1231
|
+
if (ftruncate(descriptor, (off_t)store->segment_size) != 0) {
|
|
1232
|
+
int saved_errno = errno;
|
|
1233
|
+
(void)close(descriptor);
|
|
1234
|
+
(void)unlink(temporary_path);
|
|
1235
|
+
free(path);
|
|
1236
|
+
free(temporary_path);
|
|
1237
|
+
return set_error(error,
|
|
1238
|
+
SFS_ERR_IO,
|
|
1239
|
+
saved_errno,
|
|
1240
|
+
"cannot size segment %" PRIu64 ": %s",
|
|
1241
|
+
segment_id,
|
|
1242
|
+
strerror(saved_errno));
|
|
1243
|
+
}
|
|
1244
|
+
mapping = (uint8_t *)mmap(NULL,
|
|
1245
|
+
(size_t)store->segment_size,
|
|
1246
|
+
PROT_READ | PROT_WRITE,
|
|
1247
|
+
MAP_SHARED,
|
|
1248
|
+
descriptor,
|
|
1249
|
+
0);
|
|
1250
|
+
if (mapping == MAP_FAILED) {
|
|
1251
|
+
int saved_errno = errno;
|
|
1252
|
+
(void)close(descriptor);
|
|
1253
|
+
(void)unlink(temporary_path);
|
|
1254
|
+
free(path);
|
|
1255
|
+
free(temporary_path);
|
|
1256
|
+
return set_error(error,
|
|
1257
|
+
SFS_ERR_IO,
|
|
1258
|
+
saved_errno,
|
|
1259
|
+
"cannot map new segment: %s",
|
|
1260
|
+
strerror(saved_errno));
|
|
1261
|
+
}
|
|
1262
|
+
encode_segment_header(mapping,
|
|
1263
|
+
partition_id,
|
|
1264
|
+
segment_id,
|
|
1265
|
+
store->segment_size,
|
|
1266
|
+
store->next_sequence);
|
|
1267
|
+
if (msync(mapping, SFS_SEGMENT_HEADER_SIZE, MS_SYNC) != 0 ||
|
|
1268
|
+
fsync(descriptor) != 0) {
|
|
1269
|
+
int saved_errno = errno;
|
|
1270
|
+
(void)munmap(mapping, (size_t)store->segment_size);
|
|
1271
|
+
(void)close(descriptor);
|
|
1272
|
+
(void)unlink(temporary_path);
|
|
1273
|
+
free(path);
|
|
1274
|
+
free(temporary_path);
|
|
1275
|
+
return set_error(error,
|
|
1276
|
+
SFS_ERR_IO,
|
|
1277
|
+
saved_errno,
|
|
1278
|
+
"cannot commit new segment header: %s",
|
|
1279
|
+
strerror(saved_errno));
|
|
1280
|
+
}
|
|
1281
|
+
{
|
|
1282
|
+
struct stat existing;
|
|
1283
|
+
int existing_result = lstat(path, &existing);
|
|
1284
|
+
if (existing_result == 0 || errno != ENOENT) {
|
|
1285
|
+
int saved_errno = existing_result == 0 ? EEXIST : errno;
|
|
1286
|
+
(void)munmap(mapping, (size_t)store->segment_size);
|
|
1287
|
+
(void)close(descriptor);
|
|
1288
|
+
(void)unlink(temporary_path);
|
|
1289
|
+
free(path);
|
|
1290
|
+
free(temporary_path);
|
|
1291
|
+
return set_error(error,
|
|
1292
|
+
SFS_ERR_CORRUPT,
|
|
1293
|
+
saved_errno,
|
|
1294
|
+
"segment %" PRIu64 " already exists",
|
|
1295
|
+
segment_id);
|
|
1296
|
+
}
|
|
1297
|
+
}
|
|
1298
|
+
if (rename(temporary_path, path) != 0) {
|
|
1299
|
+
int saved_errno = errno;
|
|
1300
|
+
(void)munmap(mapping, (size_t)store->segment_size);
|
|
1301
|
+
(void)close(descriptor);
|
|
1302
|
+
(void)unlink(temporary_path);
|
|
1303
|
+
free(path);
|
|
1304
|
+
free(temporary_path);
|
|
1305
|
+
return set_error(error,
|
|
1306
|
+
SFS_ERR_IO,
|
|
1307
|
+
saved_errno,
|
|
1308
|
+
"cannot publish segment %" PRIu64 ": %s",
|
|
1309
|
+
segment_id,
|
|
1310
|
+
strerror(saved_errno));
|
|
1311
|
+
}
|
|
1312
|
+
fsync_directory_best_effort(partition->directory);
|
|
1313
|
+
free(path);
|
|
1314
|
+
free(temporary_path);
|
|
1315
|
+
meta = &partition->segments[partition->segment_count++];
|
|
1316
|
+
memset(meta, 0, sizeof(*meta));
|
|
1317
|
+
meta->id = segment_id;
|
|
1318
|
+
meta->write_offset = SFS_SEGMENT_HEADER_SIZE;
|
|
1319
|
+
partition->active_fd = descriptor;
|
|
1320
|
+
partition->active_map = mapping;
|
|
1321
|
+
partition->active_write_offset = SFS_SEGMENT_HEADER_SIZE;
|
|
1322
|
+
partition->durable_offset = SFS_SEGMENT_HEADER_SIZE;
|
|
1323
|
+
partition->next_segment_id += 1u;
|
|
1324
|
+
if (partition->segment_count == 1u) {
|
|
1325
|
+
partition->first_retained_segment_id = segment_id;
|
|
1326
|
+
}
|
|
1327
|
+
return manifest_save(store, error);
|
|
1328
|
+
}
|
|
1329
|
+
|
|
1330
|
+
static sfs_result_t load_partition(sfs_store_t *store,
|
|
1331
|
+
uint32_t partition_id,
|
|
1332
|
+
bool create,
|
|
1333
|
+
sfs_error_t *error)
|
|
1334
|
+
{
|
|
1335
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1336
|
+
uint64_t *ids = NULL;
|
|
1337
|
+
size_t id_count = 0u;
|
|
1338
|
+
size_t index;
|
|
1339
|
+
sfs_result_t result;
|
|
1340
|
+
partition->directory = partition_path(store->directory, partition_id);
|
|
1341
|
+
if (partition->directory == NULL) {
|
|
1342
|
+
return set_error(error,
|
|
1343
|
+
SFS_ERR_NOMEM,
|
|
1344
|
+
errno,
|
|
1345
|
+
"cannot allocate partition directory path");
|
|
1346
|
+
}
|
|
1347
|
+
result = ensure_directory(partition->directory, create, error);
|
|
1348
|
+
if (result != SFS_OK) {
|
|
1349
|
+
return result;
|
|
1350
|
+
}
|
|
1351
|
+
result = cleanup_temporary_segments(partition, error);
|
|
1352
|
+
if (result != SFS_OK) {
|
|
1353
|
+
return result;
|
|
1354
|
+
}
|
|
1355
|
+
result = enumerate_segment_ids(partition, &ids, &id_count, error);
|
|
1356
|
+
if (result != SFS_OK) {
|
|
1357
|
+
return result;
|
|
1358
|
+
}
|
|
1359
|
+
for (index = 0u; index < id_count; ++index) {
|
|
1360
|
+
if (ids[index] < partition->first_retained_segment_id) {
|
|
1361
|
+
char *obsolete = segment_path(partition, ids[index]);
|
|
1362
|
+
if (obsolete == NULL || unlink(obsolete) != 0) {
|
|
1363
|
+
int saved_errno = errno;
|
|
1364
|
+
free(obsolete);
|
|
1365
|
+
free(ids);
|
|
1366
|
+
return set_error(error,
|
|
1367
|
+
SFS_ERR_IO,
|
|
1368
|
+
saved_errno,
|
|
1369
|
+
"cannot finish eviction of segment %" PRIu64,
|
|
1370
|
+
ids[index]);
|
|
1371
|
+
}
|
|
1372
|
+
free(obsolete);
|
|
1373
|
+
memmove(&ids[index],
|
|
1374
|
+
&ids[index + 1u],
|
|
1375
|
+
(id_count - index - 1u) * sizeof(*ids));
|
|
1376
|
+
--id_count;
|
|
1377
|
+
--index;
|
|
1378
|
+
}
|
|
1379
|
+
}
|
|
1380
|
+
if (id_count == 0u) {
|
|
1381
|
+
free(ids);
|
|
1382
|
+
return create_segment(store, partition_id, error);
|
|
1383
|
+
}
|
|
1384
|
+
if (ids[0] != partition->first_retained_segment_id) {
|
|
1385
|
+
uint64_t first_id = ids[0];
|
|
1386
|
+
free(ids);
|
|
1387
|
+
return set_error(error,
|
|
1388
|
+
SFS_ERR_CORRUPT,
|
|
1389
|
+
0,
|
|
1390
|
+
"partition %u is missing retained segment %" PRIu64,
|
|
1391
|
+
partition_id,
|
|
1392
|
+
first_id);
|
|
1393
|
+
}
|
|
1394
|
+
for (index = 1u; index < id_count; ++index) {
|
|
1395
|
+
if (ids[index] != ids[index - 1u] + 1u) {
|
|
1396
|
+
free(ids);
|
|
1397
|
+
return set_error(error,
|
|
1398
|
+
SFS_ERR_CORRUPT,
|
|
1399
|
+
0,
|
|
1400
|
+
"partition %u has a segment id gap",
|
|
1401
|
+
partition_id);
|
|
1402
|
+
}
|
|
1403
|
+
}
|
|
1404
|
+
if (id_count > partition->quota_bytes / store->segment_size) {
|
|
1405
|
+
free(ids);
|
|
1406
|
+
return set_error(error,
|
|
1407
|
+
SFS_ERR_CORRUPT,
|
|
1408
|
+
0,
|
|
1409
|
+
"partition %u exceeds its hard quota",
|
|
1410
|
+
partition_id);
|
|
1411
|
+
}
|
|
1412
|
+
result = reserve_segments(partition, id_count, error);
|
|
1413
|
+
if (result != SFS_OK) {
|
|
1414
|
+
free(ids);
|
|
1415
|
+
return result;
|
|
1416
|
+
}
|
|
1417
|
+
for (index = 0u; index < id_count; ++index) {
|
|
1418
|
+
sfs_segment_meta_t meta;
|
|
1419
|
+
bool active = index + 1u == id_count;
|
|
1420
|
+
result = scan_segment_file(store,
|
|
1421
|
+
partition_id,
|
|
1422
|
+
ids[index],
|
|
1423
|
+
active,
|
|
1424
|
+
&meta,
|
|
1425
|
+
error);
|
|
1426
|
+
if (result != SFS_OK) {
|
|
1427
|
+
free(ids);
|
|
1428
|
+
return result;
|
|
1429
|
+
}
|
|
1430
|
+
partition->segments[partition->segment_count++] = meta;
|
|
1431
|
+
partition->record_count += meta.record_count;
|
|
1432
|
+
partition->payload_bytes += meta.payload_bytes;
|
|
1433
|
+
if (meta.record_count != 0u) {
|
|
1434
|
+
if (partition->first_sequence == 0u) {
|
|
1435
|
+
partition->first_sequence = meta.first_sequence;
|
|
1436
|
+
}
|
|
1437
|
+
partition->last_sequence = meta.last_sequence;
|
|
1438
|
+
if (meta.last_sequence >= store->next_sequence) {
|
|
1439
|
+
if (meta.last_sequence == UINT64_MAX) {
|
|
1440
|
+
free(ids);
|
|
1441
|
+
return set_error(error,
|
|
1442
|
+
SFS_ERR_FULL,
|
|
1443
|
+
0,
|
|
1444
|
+
"global sequence is exhausted");
|
|
1445
|
+
}
|
|
1446
|
+
store->next_sequence = meta.last_sequence + 1u;
|
|
1447
|
+
}
|
|
1448
|
+
}
|
|
1449
|
+
}
|
|
1450
|
+
if (ids[id_count - 1u] >= partition->next_segment_id) {
|
|
1451
|
+
if (ids[id_count - 1u] == UINT64_MAX) {
|
|
1452
|
+
free(ids);
|
|
1453
|
+
return set_error(error, SFS_ERR_FULL, 0, "segment id is exhausted");
|
|
1454
|
+
}
|
|
1455
|
+
partition->next_segment_id = ids[id_count - 1u] + 1u;
|
|
1456
|
+
}
|
|
1457
|
+
free(ids);
|
|
1458
|
+
return SFS_OK;
|
|
1459
|
+
}
|
|
1460
|
+
|
|
1461
|
+
static sfs_result_t flush_partition(sfs_store_t *store,
|
|
1462
|
+
sfs_partition_t *partition,
|
|
1463
|
+
sfs_error_t *error)
|
|
1464
|
+
{
|
|
1465
|
+
uint64_t offset;
|
|
1466
|
+
int first_error = 0;
|
|
1467
|
+
if (partition->active_map == NULL ||
|
|
1468
|
+
partition->durable_offset >= partition->active_write_offset) {
|
|
1469
|
+
return SFS_OK;
|
|
1470
|
+
}
|
|
1471
|
+
offset = partition->durable_offset;
|
|
1472
|
+
while (offset < partition->active_write_offset) {
|
|
1473
|
+
uint32_t total_length = load_u32_le(partition->active_map + offset);
|
|
1474
|
+
uint64_t marker_offset;
|
|
1475
|
+
if (total_length < SFS_MIN_FRAME_SIZE || (total_length & 7u) != 0u ||
|
|
1476
|
+
total_length > partition->active_write_offset - offset) {
|
|
1477
|
+
return set_error(error,
|
|
1478
|
+
SFS_ERR_CORRUPT,
|
|
1479
|
+
0,
|
|
1480
|
+
"active frame layout is corrupt during flush");
|
|
1481
|
+
}
|
|
1482
|
+
marker_offset = offset + total_length - SFS_FRAME_COMMIT_SIZE;
|
|
1483
|
+
store_u64_le(partition->active_map + marker_offset, 0u);
|
|
1484
|
+
offset += total_length;
|
|
1485
|
+
}
|
|
1486
|
+
atomic_thread_fence(memory_order_release);
|
|
1487
|
+
if (msync(partition->active_map, (size_t)store->segment_size, MS_SYNC) != 0) {
|
|
1488
|
+
first_error = errno;
|
|
1489
|
+
}
|
|
1490
|
+
offset = partition->durable_offset;
|
|
1491
|
+
while (offset < partition->active_write_offset) {
|
|
1492
|
+
uint32_t total_length = load_u32_le(partition->active_map + offset);
|
|
1493
|
+
uint64_t marker_offset = offset + total_length - SFS_FRAME_COMMIT_SIZE;
|
|
1494
|
+
store_u64_le(partition->active_map + marker_offset,
|
|
1495
|
+
SFS_FRAME_COMMIT_MARKER);
|
|
1496
|
+
offset += total_length;
|
|
1497
|
+
}
|
|
1498
|
+
atomic_thread_fence(memory_order_release);
|
|
1499
|
+
if (first_error == 0 &&
|
|
1500
|
+
(msync(partition->active_map,
|
|
1501
|
+
(size_t)store->segment_size,
|
|
1502
|
+
MS_SYNC) != 0 ||
|
|
1503
|
+
fsync(partition->active_fd) != 0)) {
|
|
1504
|
+
first_error = errno;
|
|
1505
|
+
}
|
|
1506
|
+
if (first_error != 0) {
|
|
1507
|
+
return set_error(error,
|
|
1508
|
+
SFS_ERR_IO,
|
|
1509
|
+
first_error,
|
|
1510
|
+
"cannot flush segment: %s",
|
|
1511
|
+
strerror(first_error));
|
|
1512
|
+
}
|
|
1513
|
+
partition->durable_offset = partition->active_write_offset;
|
|
1514
|
+
return SFS_OK;
|
|
1515
|
+
}
|
|
1516
|
+
|
|
1517
|
+
static void close_active_segment(sfs_store_t *store, sfs_partition_t *partition)
|
|
1518
|
+
{
|
|
1519
|
+
if (partition->active_map != NULL) {
|
|
1520
|
+
(void)munmap(partition->active_map, (size_t)store->segment_size);
|
|
1521
|
+
partition->active_map = NULL;
|
|
1522
|
+
}
|
|
1523
|
+
if (partition->active_fd >= 0) {
|
|
1524
|
+
(void)close(partition->active_fd);
|
|
1525
|
+
partition->active_fd = -1;
|
|
1526
|
+
}
|
|
1527
|
+
}
|
|
1528
|
+
|
|
1529
|
+
static sfs_result_t evict_oldest_segment(sfs_store_t *store,
|
|
1530
|
+
uint32_t partition_id,
|
|
1531
|
+
sfs_error_t *error)
|
|
1532
|
+
{
|
|
1533
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1534
|
+
sfs_segment_meta_t victim;
|
|
1535
|
+
char *path;
|
|
1536
|
+
sfs_result_t result;
|
|
1537
|
+
if (partition->segment_count == 0u) {
|
|
1538
|
+
return set_error(error, SFS_ERR_CORRUPT, 0, "no segment is available to evict");
|
|
1539
|
+
}
|
|
1540
|
+
victim = partition->segments[0];
|
|
1541
|
+
partition->evicted_segments += 1u;
|
|
1542
|
+
partition->evicted_records += victim.record_count;
|
|
1543
|
+
partition->evicted_payload_bytes += victim.payload_bytes;
|
|
1544
|
+
partition->first_retained_segment_id = victim.id + 1u;
|
|
1545
|
+
result = manifest_save(store, error);
|
|
1546
|
+
if (result != SFS_OK) {
|
|
1547
|
+
partition->evicted_segments -= 1u;
|
|
1548
|
+
partition->evicted_records -= victim.record_count;
|
|
1549
|
+
partition->evicted_payload_bytes -= victim.payload_bytes;
|
|
1550
|
+
partition->first_retained_segment_id = victim.id;
|
|
1551
|
+
return result;
|
|
1552
|
+
}
|
|
1553
|
+
path = segment_path(partition, victim.id);
|
|
1554
|
+
if (path == NULL || unlink(path) != 0) {
|
|
1555
|
+
int saved_errno = errno;
|
|
1556
|
+
free(path);
|
|
1557
|
+
return set_error(error,
|
|
1558
|
+
SFS_ERR_IO,
|
|
1559
|
+
saved_errno,
|
|
1560
|
+
"cannot evict segment %" PRIu64 ": %s",
|
|
1561
|
+
victim.id,
|
|
1562
|
+
strerror(saved_errno));
|
|
1563
|
+
}
|
|
1564
|
+
free(path);
|
|
1565
|
+
fsync_directory_best_effort(partition->directory);
|
|
1566
|
+
partition->record_count -= victim.record_count;
|
|
1567
|
+
partition->payload_bytes -= victim.payload_bytes;
|
|
1568
|
+
memmove(&partition->segments[0],
|
|
1569
|
+
&partition->segments[1],
|
|
1570
|
+
(partition->segment_count - 1u) * sizeof(*partition->segments));
|
|
1571
|
+
partition->segment_count -= 1u;
|
|
1572
|
+
partition->first_sequence = partition->segment_count == 0u
|
|
1573
|
+
? 0u
|
|
1574
|
+
: partition->segments[0].first_sequence;
|
|
1575
|
+
partition->last_sequence = partition->segment_count == 0u
|
|
1576
|
+
? 0u
|
|
1577
|
+
: partition->segments[partition->segment_count - 1u]
|
|
1578
|
+
.last_sequence;
|
|
1579
|
+
return SFS_OK;
|
|
1580
|
+
}
|
|
1581
|
+
|
|
1582
|
+
static sfs_result_t rotate_partition(sfs_store_t *store,
|
|
1583
|
+
uint32_t partition_id,
|
|
1584
|
+
sfs_error_t *error)
|
|
1585
|
+
{
|
|
1586
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1587
|
+
sfs_result_t result = flush_partition(store, partition, error);
|
|
1588
|
+
if (result != SFS_OK) {
|
|
1589
|
+
return result;
|
|
1590
|
+
}
|
|
1591
|
+
close_active_segment(store, partition);
|
|
1592
|
+
while (partition->segment_count >=
|
|
1593
|
+
partition->quota_bytes / store->segment_size) {
|
|
1594
|
+
result = evict_oldest_segment(store, partition_id, error);
|
|
1595
|
+
if (result != SFS_OK) {
|
|
1596
|
+
return result;
|
|
1597
|
+
}
|
|
1598
|
+
}
|
|
1599
|
+
return create_segment(store, partition_id, error);
|
|
1600
|
+
}
|
|
1601
|
+
|
|
1602
|
+
static sfs_result_t flush_locked(sfs_store_t *store, sfs_error_t *error)
|
|
1603
|
+
{
|
|
1604
|
+
uint32_t partition_id;
|
|
1605
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
1606
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1607
|
+
sfs_result_t result;
|
|
1608
|
+
if (!partition->enabled) {
|
|
1609
|
+
continue;
|
|
1610
|
+
}
|
|
1611
|
+
result = flush_partition(store, partition, error);
|
|
1612
|
+
if (result != SFS_OK) {
|
|
1613
|
+
return result;
|
|
1614
|
+
}
|
|
1615
|
+
}
|
|
1616
|
+
return manifest_save(store, error);
|
|
1617
|
+
}
|
|
1618
|
+
|
|
1619
|
+
static sfs_result_t map_segment_for_read(sfs_store_t *store,
|
|
1620
|
+
uint32_t partition_id,
|
|
1621
|
+
size_t segment_index,
|
|
1622
|
+
const uint8_t **out_mapping,
|
|
1623
|
+
int *out_descriptor,
|
|
1624
|
+
bool *out_borrowed,
|
|
1625
|
+
sfs_error_t *error)
|
|
1626
|
+
{
|
|
1627
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1628
|
+
sfs_segment_meta_t *meta = &partition->segments[segment_index];
|
|
1629
|
+
if (segment_index + 1u == partition->segment_count &&
|
|
1630
|
+
partition->active_map != NULL) {
|
|
1631
|
+
*out_mapping = partition->active_map;
|
|
1632
|
+
*out_descriptor = -1;
|
|
1633
|
+
*out_borrowed = true;
|
|
1634
|
+
return SFS_OK;
|
|
1635
|
+
}
|
|
1636
|
+
{
|
|
1637
|
+
char *path = segment_path(partition, meta->id);
|
|
1638
|
+
int descriptor;
|
|
1639
|
+
uint8_t *mapping;
|
|
1640
|
+
if (path == NULL) {
|
|
1641
|
+
return set_error(error,
|
|
1642
|
+
SFS_ERR_NOMEM,
|
|
1643
|
+
errno,
|
|
1644
|
+
"cannot allocate scan path");
|
|
1645
|
+
}
|
|
1646
|
+
descriptor = open(path, O_RDONLY);
|
|
1647
|
+
free(path);
|
|
1648
|
+
if (descriptor < 0) {
|
|
1649
|
+
int saved_errno = errno;
|
|
1650
|
+
return set_error(error,
|
|
1651
|
+
SFS_ERR_IO,
|
|
1652
|
+
saved_errno,
|
|
1653
|
+
"cannot open segment for scan: %s",
|
|
1654
|
+
strerror(saved_errno));
|
|
1655
|
+
}
|
|
1656
|
+
mapping = (uint8_t *)mmap(NULL,
|
|
1657
|
+
(size_t)store->segment_size,
|
|
1658
|
+
PROT_READ,
|
|
1659
|
+
MAP_SHARED,
|
|
1660
|
+
descriptor,
|
|
1661
|
+
0);
|
|
1662
|
+
if (mapping == MAP_FAILED) {
|
|
1663
|
+
int saved_errno = errno;
|
|
1664
|
+
(void)close(descriptor);
|
|
1665
|
+
return set_error(error,
|
|
1666
|
+
SFS_ERR_IO,
|
|
1667
|
+
saved_errno,
|
|
1668
|
+
"cannot map segment for scan: %s",
|
|
1669
|
+
strerror(saved_errno));
|
|
1670
|
+
}
|
|
1671
|
+
*out_mapping = mapping;
|
|
1672
|
+
*out_descriptor = descriptor;
|
|
1673
|
+
*out_borrowed = false;
|
|
1674
|
+
}
|
|
1675
|
+
return SFS_OK;
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
static void unmap_segment_for_read(sfs_store_t *store,
|
|
1679
|
+
const uint8_t *mapping,
|
|
1680
|
+
int descriptor,
|
|
1681
|
+
bool borrowed)
|
|
1682
|
+
{
|
|
1683
|
+
if (!borrowed) {
|
|
1684
|
+
(void)munmap((void *)mapping, (size_t)store->segment_size);
|
|
1685
|
+
(void)close(descriptor);
|
|
1686
|
+
}
|
|
1687
|
+
}
|
|
1688
|
+
|
|
1689
|
+
static sfs_result_t candidate_in_partition(sfs_store_t *store,
|
|
1690
|
+
uint32_t partition_id,
|
|
1691
|
+
uint64_t after_sequence,
|
|
1692
|
+
sfs_candidate_t *candidate,
|
|
1693
|
+
sfs_error_t *error)
|
|
1694
|
+
{
|
|
1695
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1696
|
+
size_t segment_index;
|
|
1697
|
+
bool resume = partition->scan_segment_id != 0u &&
|
|
1698
|
+
after_sequence >= partition->scan_after_sequence;
|
|
1699
|
+
memset(candidate, 0, sizeof(*candidate));
|
|
1700
|
+
for (segment_index = 0u; segment_index < partition->segment_count;
|
|
1701
|
+
++segment_index) {
|
|
1702
|
+
sfs_segment_meta_t *meta = &partition->segments[segment_index];
|
|
1703
|
+
const uint8_t *mapping;
|
|
1704
|
+
int descriptor;
|
|
1705
|
+
bool borrowed;
|
|
1706
|
+
uint64_t offset = SFS_SEGMENT_HEADER_SIZE;
|
|
1707
|
+
sfs_result_t result;
|
|
1708
|
+
if (meta->record_count == 0u || meta->last_sequence <= after_sequence) {
|
|
1709
|
+
continue;
|
|
1710
|
+
}
|
|
1711
|
+
if (resume && meta->id == partition->scan_segment_id) {
|
|
1712
|
+
offset = partition->scan_offset;
|
|
1713
|
+
}
|
|
1714
|
+
result = map_segment_for_read(store,
|
|
1715
|
+
partition_id,
|
|
1716
|
+
segment_index,
|
|
1717
|
+
&mapping,
|
|
1718
|
+
&descriptor,
|
|
1719
|
+
&borrowed,
|
|
1720
|
+
error);
|
|
1721
|
+
if (result != SFS_OK) {
|
|
1722
|
+
return result;
|
|
1723
|
+
}
|
|
1724
|
+
while (offset < meta->write_offset) {
|
|
1725
|
+
sfs_frame_view_t frame;
|
|
1726
|
+
int decoded = decode_frame(mapping,
|
|
1727
|
+
store->segment_size,
|
|
1728
|
+
store->segment_size,
|
|
1729
|
+
offset,
|
|
1730
|
+
&frame,
|
|
1731
|
+
error);
|
|
1732
|
+
if (decoded != SFS_OK) {
|
|
1733
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1734
|
+
return decoded == SFS_END || decoded == SFS_INTERNAL_TORN
|
|
1735
|
+
? set_error(error,
|
|
1736
|
+
SFS_ERR_CORRUPT,
|
|
1737
|
+
0,
|
|
1738
|
+
"retained frame disappeared during scan")
|
|
1739
|
+
: (sfs_result_t)decoded;
|
|
1740
|
+
}
|
|
1741
|
+
if (frame.sequence > after_sequence) {
|
|
1742
|
+
partition->scan_after_sequence = after_sequence;
|
|
1743
|
+
partition->scan_segment_id = meta->id;
|
|
1744
|
+
partition->scan_offset = offset;
|
|
1745
|
+
candidate->present = true;
|
|
1746
|
+
candidate->partition_id = partition_id;
|
|
1747
|
+
candidate->segment_index = segment_index;
|
|
1748
|
+
candidate->frame_offset = offset;
|
|
1749
|
+
candidate->frame = frame;
|
|
1750
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1751
|
+
return SFS_OK;
|
|
1752
|
+
}
|
|
1753
|
+
offset += frame.total_length;
|
|
1754
|
+
}
|
|
1755
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1756
|
+
}
|
|
1757
|
+
return SFS_OK;
|
|
1758
|
+
}
|
|
1759
|
+
|
|
1760
|
+
static sfs_result_t copy_candidate(sfs_store_t *store,
|
|
1761
|
+
const sfs_candidate_t *candidate,
|
|
1762
|
+
void *buffer,
|
|
1763
|
+
uint32_t buffer_capacity,
|
|
1764
|
+
sfs_record_info_t *out_record,
|
|
1765
|
+
sfs_error_t *error)
|
|
1766
|
+
{
|
|
1767
|
+
const uint8_t *mapping;
|
|
1768
|
+
int descriptor;
|
|
1769
|
+
bool borrowed;
|
|
1770
|
+
sfs_frame_view_t frame;
|
|
1771
|
+
int decoded;
|
|
1772
|
+
sfs_result_t result = map_segment_for_read(store,
|
|
1773
|
+
candidate->partition_id,
|
|
1774
|
+
candidate->segment_index,
|
|
1775
|
+
&mapping,
|
|
1776
|
+
&descriptor,
|
|
1777
|
+
&borrowed,
|
|
1778
|
+
error);
|
|
1779
|
+
if (result != SFS_OK) {
|
|
1780
|
+
return result;
|
|
1781
|
+
}
|
|
1782
|
+
decoded = decode_frame(mapping,
|
|
1783
|
+
store->segment_size,
|
|
1784
|
+
store->segment_size,
|
|
1785
|
+
candidate->frame_offset,
|
|
1786
|
+
&frame,
|
|
1787
|
+
error);
|
|
1788
|
+
if (decoded != SFS_OK) {
|
|
1789
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1790
|
+
return decoded == SFS_END || decoded == SFS_INTERNAL_TORN
|
|
1791
|
+
? set_error(error,
|
|
1792
|
+
SFS_ERR_CORRUPT,
|
|
1793
|
+
0,
|
|
1794
|
+
"retained frame disappeared during scan")
|
|
1795
|
+
: (sfs_result_t)decoded;
|
|
1796
|
+
}
|
|
1797
|
+
memset(out_record, 0, sizeof(*out_record));
|
|
1798
|
+
out_record->struct_size = sizeof(*out_record);
|
|
1799
|
+
out_record->payload_length = frame.payload_length;
|
|
1800
|
+
out_record->partition_id = candidate->partition_id;
|
|
1801
|
+
out_record->sequence = frame.sequence;
|
|
1802
|
+
out_record->segment_id =
|
|
1803
|
+
store->partitions[candidate->partition_id]
|
|
1804
|
+
.segments[candidate->segment_index]
|
|
1805
|
+
.id;
|
|
1806
|
+
out_record->frame_offset = candidate->frame_offset;
|
|
1807
|
+
if (buffer_capacity < frame.payload_length) {
|
|
1808
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1809
|
+
return SFS_BUFFER_TOO_SMALL;
|
|
1810
|
+
}
|
|
1811
|
+
if (frame.payload_length != 0u) {
|
|
1812
|
+
memcpy(buffer, mapping + frame.payload_offset, frame.payload_length);
|
|
1813
|
+
}
|
|
1814
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1815
|
+
return SFS_OK;
|
|
1816
|
+
}
|
|
1817
|
+
|
|
1818
|
+
static size_t find_segment_index(const sfs_partition_t *partition,
|
|
1819
|
+
uint64_t segment_id)
|
|
1820
|
+
{
|
|
1821
|
+
size_t index;
|
|
1822
|
+
for (index = 0u; index < partition->segment_count; ++index) {
|
|
1823
|
+
if (partition->segments[index].id >= segment_id) {
|
|
1824
|
+
return index;
|
|
1825
|
+
}
|
|
1826
|
+
}
|
|
1827
|
+
return partition->segment_count;
|
|
1828
|
+
}
|
|
1829
|
+
|
|
1830
|
+
static sfs_result_t scan_partition_cursor(sfs_store_t *store,
|
|
1831
|
+
sfs_cursor_t *cursor,
|
|
1832
|
+
void *buffer,
|
|
1833
|
+
uint32_t buffer_capacity,
|
|
1834
|
+
sfs_record_info_t *out_record,
|
|
1835
|
+
sfs_error_t *error)
|
|
1836
|
+
{
|
|
1837
|
+
sfs_partition_t *partition = &store->partitions[cursor->partition_id];
|
|
1838
|
+
size_t segment_index;
|
|
1839
|
+
uint64_t offset;
|
|
1840
|
+
bool cursor_segment_evicted = false;
|
|
1841
|
+
if (!partition->enabled) {
|
|
1842
|
+
return set_error(error,
|
|
1843
|
+
SFS_ERR_PARTITION_DISABLED,
|
|
1844
|
+
0,
|
|
1845
|
+
"partition %u is disabled",
|
|
1846
|
+
cursor->partition_id);
|
|
1847
|
+
}
|
|
1848
|
+
segment_index = cursor->segment_id == 0u
|
|
1849
|
+
? 0u
|
|
1850
|
+
: find_segment_index(partition, cursor->segment_id);
|
|
1851
|
+
if (cursor->segment_id == 0u) {
|
|
1852
|
+
cursor_segment_evicted = partition->evicted_records != 0u;
|
|
1853
|
+
} else if (segment_index < partition->segment_count &&
|
|
1854
|
+
partition->segments[segment_index].id > cursor->segment_id) {
|
|
1855
|
+
cursor_segment_evicted = true;
|
|
1856
|
+
}
|
|
1857
|
+
offset = cursor->offset == 0u ? SFS_SEGMENT_HEADER_SIZE : cursor->offset;
|
|
1858
|
+
for (; segment_index < partition->segment_count; ++segment_index) {
|
|
1859
|
+
sfs_segment_meta_t *meta = &partition->segments[segment_index];
|
|
1860
|
+
const uint8_t *mapping;
|
|
1861
|
+
int descriptor;
|
|
1862
|
+
bool borrowed;
|
|
1863
|
+
sfs_frame_view_t frame;
|
|
1864
|
+
sfs_candidate_t candidate;
|
|
1865
|
+
int decoded;
|
|
1866
|
+
sfs_result_t result;
|
|
1867
|
+
if (cursor->segment_id != meta->id) {
|
|
1868
|
+
offset = SFS_SEGMENT_HEADER_SIZE;
|
|
1869
|
+
}
|
|
1870
|
+
if (offset >= meta->write_offset) {
|
|
1871
|
+
cursor->segment_id = meta->id;
|
|
1872
|
+
cursor->offset = meta->write_offset;
|
|
1873
|
+
continue;
|
|
1874
|
+
}
|
|
1875
|
+
result = map_segment_for_read(store,
|
|
1876
|
+
cursor->partition_id,
|
|
1877
|
+
segment_index,
|
|
1878
|
+
&mapping,
|
|
1879
|
+
&descriptor,
|
|
1880
|
+
&borrowed,
|
|
1881
|
+
error);
|
|
1882
|
+
if (result != SFS_OK) {
|
|
1883
|
+
return result;
|
|
1884
|
+
}
|
|
1885
|
+
decoded = decode_frame(mapping,
|
|
1886
|
+
store->segment_size,
|
|
1887
|
+
store->segment_size,
|
|
1888
|
+
offset,
|
|
1889
|
+
&frame,
|
|
1890
|
+
error);
|
|
1891
|
+
unmap_segment_for_read(store, mapping, descriptor, borrowed);
|
|
1892
|
+
if (decoded != SFS_OK) {
|
|
1893
|
+
return decoded == SFS_END || decoded == SFS_INTERNAL_TORN
|
|
1894
|
+
? set_error(error,
|
|
1895
|
+
SFS_ERR_CORRUPT,
|
|
1896
|
+
0,
|
|
1897
|
+
"retained frame disappeared during scan")
|
|
1898
|
+
: (sfs_result_t)decoded;
|
|
1899
|
+
}
|
|
1900
|
+
memset(&candidate, 0, sizeof(candidate));
|
|
1901
|
+
candidate.present = true;
|
|
1902
|
+
candidate.partition_id = cursor->partition_id;
|
|
1903
|
+
candidate.segment_index = segment_index;
|
|
1904
|
+
candidate.frame_offset = offset;
|
|
1905
|
+
candidate.frame = frame;
|
|
1906
|
+
result = copy_candidate(store,
|
|
1907
|
+
&candidate,
|
|
1908
|
+
buffer,
|
|
1909
|
+
buffer_capacity,
|
|
1910
|
+
out_record,
|
|
1911
|
+
error);
|
|
1912
|
+
if (result == SFS_OK) {
|
|
1913
|
+
if (cursor_segment_evicted &&
|
|
1914
|
+
cursor->after_sequence != UINT64_MAX &&
|
|
1915
|
+
frame.sequence > cursor->after_sequence + 1u) {
|
|
1916
|
+
out_record->flags |= SFS_RECORD_GAP_BEFORE;
|
|
1917
|
+
out_record->gap_first_sequence = cursor->after_sequence + 1u;
|
|
1918
|
+
out_record->gap_last_sequence = frame.sequence - 1u;
|
|
1919
|
+
}
|
|
1920
|
+
cursor->segment_id = meta->id;
|
|
1921
|
+
cursor->offset = offset + frame.total_length;
|
|
1922
|
+
cursor->after_sequence = frame.sequence;
|
|
1923
|
+
}
|
|
1924
|
+
return result;
|
|
1925
|
+
}
|
|
1926
|
+
return SFS_END;
|
|
1927
|
+
}
|
|
1928
|
+
|
|
1929
|
+
/* A sequence-only partition cursor seeks once, then becomes an ordinary
|
|
1930
|
+
physical cursor. The hint only avoids re-decoding excluded prefixes. */
|
|
1931
|
+
static sfs_result_t seek_partition_cursor(sfs_store_t *store,
|
|
1932
|
+
sfs_cursor_t *cursor,
|
|
1933
|
+
void *buffer,
|
|
1934
|
+
uint32_t buffer_capacity,
|
|
1935
|
+
sfs_record_info_t *out_record,
|
|
1936
|
+
sfs_error_t *error)
|
|
1937
|
+
{
|
|
1938
|
+
sfs_partition_t *partition = &store->partitions[cursor->partition_id];
|
|
1939
|
+
sfs_candidate_t candidate;
|
|
1940
|
+
sfs_result_t result;
|
|
1941
|
+
if (!partition->enabled) {
|
|
1942
|
+
return set_error(error, SFS_ERR_PARTITION_DISABLED, 0,
|
|
1943
|
+
"partition %u is disabled", cursor->partition_id);
|
|
1944
|
+
}
|
|
1945
|
+
result = candidate_in_partition(store, cursor->partition_id,
|
|
1946
|
+
cursor->after_sequence, &candidate, error);
|
|
1947
|
+
if (result != SFS_OK) { return result; }
|
|
1948
|
+
if (!candidate.present) { return SFS_END; }
|
|
1949
|
+
result = copy_candidate(store, &candidate, buffer, buffer_capacity, out_record, error);
|
|
1950
|
+
if (result == SFS_OK) {
|
|
1951
|
+
if (partition->evicted_records != 0u && cursor->after_sequence < partition->first_sequence &&
|
|
1952
|
+
candidate.frame.sequence > cursor->after_sequence + 1u) {
|
|
1953
|
+
out_record->flags |= SFS_RECORD_GAP_BEFORE;
|
|
1954
|
+
out_record->gap_first_sequence = cursor->after_sequence + 1u;
|
|
1955
|
+
out_record->gap_last_sequence = candidate.frame.sequence - 1u;
|
|
1956
|
+
}
|
|
1957
|
+
cursor->after_sequence = candidate.frame.sequence;
|
|
1958
|
+
cursor->segment_id = out_record->segment_id;
|
|
1959
|
+
cursor->offset = candidate.frame_offset + candidate.frame.total_length;
|
|
1960
|
+
}
|
|
1961
|
+
return result;
|
|
1962
|
+
}
|
|
1963
|
+
|
|
1964
|
+
static void cleanup_store(sfs_store_t *store)
|
|
1965
|
+
{
|
|
1966
|
+
uint32_t partition_id;
|
|
1967
|
+
if (store == NULL) {
|
|
1968
|
+
return;
|
|
1969
|
+
}
|
|
1970
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
1971
|
+
sfs_partition_t *partition = &store->partitions[partition_id];
|
|
1972
|
+
close_active_segment(store, partition);
|
|
1973
|
+
free(partition->segments);
|
|
1974
|
+
free(partition->directory);
|
|
1975
|
+
}
|
|
1976
|
+
if (store->manifest_map != NULL) {
|
|
1977
|
+
(void)munmap(store->manifest_map, SFS_MANIFEST_SIZE);
|
|
1978
|
+
}
|
|
1979
|
+
if (store->manifest_fd >= 0) {
|
|
1980
|
+
(void)close(store->manifest_fd);
|
|
1981
|
+
}
|
|
1982
|
+
if (store->lock_fd >= 0) {
|
|
1983
|
+
(void)close(store->lock_fd);
|
|
1984
|
+
}
|
|
1985
|
+
free(store->directory);
|
|
1986
|
+
if (store->mutex_initialized) {
|
|
1987
|
+
(void)pthread_mutex_destroy(&store->mutex);
|
|
1988
|
+
}
|
|
1989
|
+
free(store);
|
|
1990
|
+
}
|
|
1991
|
+
|
|
1992
|
+
sfs_result_t sfs_open(const sfs_open_options_t *options,
|
|
1993
|
+
sfs_store_t **out_store,
|
|
1994
|
+
sfs_error_t *error)
|
|
1995
|
+
{
|
|
1996
|
+
sfs_store_t *store;
|
|
1997
|
+
bool create;
|
|
1998
|
+
uint32_t partition_id;
|
|
1999
|
+
sfs_result_t result;
|
|
2000
|
+
clear_error(error);
|
|
2001
|
+
if (out_store != NULL) {
|
|
2002
|
+
*out_store = NULL;
|
|
2003
|
+
}
|
|
2004
|
+
if (options == NULL || out_store == NULL ||
|
|
2005
|
+
options->struct_size < sizeof(*options) || options->directory == NULL ||
|
|
2006
|
+
options->directory[0] == '\0' ||
|
|
2007
|
+
(options->flags & ~((uint32_t)SFS_OPEN_CREATE)) != 0u) {
|
|
2008
|
+
return set_error(error,
|
|
2009
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2010
|
+
0,
|
|
2011
|
+
"invalid open options");
|
|
2012
|
+
}
|
|
2013
|
+
create = (options->flags & SFS_OPEN_CREATE) != 0u;
|
|
2014
|
+
result = ensure_directory(options->directory, create, error);
|
|
2015
|
+
if (result != SFS_OK) {
|
|
2016
|
+
return result;
|
|
2017
|
+
}
|
|
2018
|
+
store = (sfs_store_t *)calloc(1u, sizeof(*store));
|
|
2019
|
+
if (store == NULL) {
|
|
2020
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot allocate store");
|
|
2021
|
+
}
|
|
2022
|
+
store->lock_fd = -1;
|
|
2023
|
+
store->manifest_fd = -1;
|
|
2024
|
+
store->manifest_active_slot = -1;
|
|
2025
|
+
store->recovery_partition_id = SFS_PARTITION_ALL;
|
|
2026
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
2027
|
+
store->partitions[partition_id].active_fd = -1;
|
|
2028
|
+
}
|
|
2029
|
+
store->directory = strdup(options->directory);
|
|
2030
|
+
if (store->directory == NULL) {
|
|
2031
|
+
cleanup_store(store);
|
|
2032
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot copy store path");
|
|
2033
|
+
}
|
|
2034
|
+
if (pthread_mutex_init(&store->mutex, NULL) != 0) {
|
|
2035
|
+
cleanup_store(store);
|
|
2036
|
+
return set_error(error, SFS_ERR_NOMEM, errno, "cannot initialize store");
|
|
2037
|
+
}
|
|
2038
|
+
store->mutex_initialized = true;
|
|
2039
|
+
result = lock_store_directory(store, error);
|
|
2040
|
+
if (result != SFS_OK) {
|
|
2041
|
+
cleanup_store(store);
|
|
2042
|
+
return result;
|
|
2043
|
+
}
|
|
2044
|
+
result = open_manifest(store, options, error);
|
|
2045
|
+
if (result != SFS_OK) {
|
|
2046
|
+
cleanup_store(store);
|
|
2047
|
+
return result;
|
|
2048
|
+
}
|
|
2049
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
2050
|
+
if (!store->partitions[partition_id].enabled) {
|
|
2051
|
+
continue;
|
|
2052
|
+
}
|
|
2053
|
+
result = load_partition(store, partition_id, create, error);
|
|
2054
|
+
if (result != SFS_OK) {
|
|
2055
|
+
cleanup_store(store);
|
|
2056
|
+
return result;
|
|
2057
|
+
}
|
|
2058
|
+
}
|
|
2059
|
+
result = manifest_save(store, error);
|
|
2060
|
+
if (result != SFS_OK) {
|
|
2061
|
+
cleanup_store(store);
|
|
2062
|
+
return result;
|
|
2063
|
+
}
|
|
2064
|
+
*out_store = store;
|
|
2065
|
+
return SFS_OK;
|
|
2066
|
+
}
|
|
2067
|
+
|
|
2068
|
+
sfs_result_t sfs_append(sfs_store_t *store,
|
|
2069
|
+
uint32_t partition_id,
|
|
2070
|
+
const void *payload,
|
|
2071
|
+
uint32_t payload_length,
|
|
2072
|
+
sfs_durability_t durability,
|
|
2073
|
+
sfs_record_info_t *out_record,
|
|
2074
|
+
sfs_error_t *error)
|
|
2075
|
+
{
|
|
2076
|
+
sfs_partition_t *partition;
|
|
2077
|
+
sfs_segment_meta_t *meta;
|
|
2078
|
+
uint64_t frame_length;
|
|
2079
|
+
uint64_t offset;
|
|
2080
|
+
uint64_t commit_offset;
|
|
2081
|
+
uint64_t sequence;
|
|
2082
|
+
sfs_result_t result = SFS_OK;
|
|
2083
|
+
clear_error(error);
|
|
2084
|
+
if (store == NULL || partition_id >= SFS_MAX_PARTITIONS ||
|
|
2085
|
+
(payload == NULL && payload_length != 0u) ||
|
|
2086
|
+
out_record == NULL || out_record->struct_size < sizeof(*out_record) ||
|
|
2087
|
+
(durability != SFS_DURABILITY_MEMORY &&
|
|
2088
|
+
durability != SFS_DURABILITY_SYNC)) {
|
|
2089
|
+
return set_error(error,
|
|
2090
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2091
|
+
0,
|
|
2092
|
+
"invalid append arguments");
|
|
2093
|
+
}
|
|
2094
|
+
memset(out_record, 0, sizeof(*out_record));
|
|
2095
|
+
out_record->struct_size = sizeof(*out_record);
|
|
2096
|
+
(void)pthread_mutex_lock(&store->mutex);
|
|
2097
|
+
partition = &store->partitions[partition_id];
|
|
2098
|
+
if (!partition->enabled) {
|
|
2099
|
+
result = set_error(error,
|
|
2100
|
+
SFS_ERR_PARTITION_DISABLED,
|
|
2101
|
+
0,
|
|
2102
|
+
"partition %u is disabled",
|
|
2103
|
+
partition_id);
|
|
2104
|
+
goto done;
|
|
2105
|
+
}
|
|
2106
|
+
frame_length = align_eight(SFS_FRAME_PREFIX_SIZE +
|
|
2107
|
+
(uint64_t)payload_length) +
|
|
2108
|
+
SFS_FRAME_COMMIT_SIZE;
|
|
2109
|
+
if (frame_length > store->segment_size - SFS_SEGMENT_HEADER_SIZE) {
|
|
2110
|
+
result = set_error(error,
|
|
2111
|
+
SFS_ERR_FULL,
|
|
2112
|
+
0,
|
|
2113
|
+
"record does not fit in an empty segment");
|
|
2114
|
+
goto done;
|
|
2115
|
+
}
|
|
2116
|
+
if (store->next_sequence == UINT64_MAX) {
|
|
2117
|
+
result = set_error(error, SFS_ERR_FULL, 0, "global sequence is exhausted");
|
|
2118
|
+
goto done;
|
|
2119
|
+
}
|
|
2120
|
+
if (partition->active_map == NULL ||
|
|
2121
|
+
partition->active_write_offset > store->segment_size - frame_length) {
|
|
2122
|
+
result = rotate_partition(store, partition_id, error);
|
|
2123
|
+
if (result != SFS_OK) {
|
|
2124
|
+
goto done;
|
|
2125
|
+
}
|
|
2126
|
+
}
|
|
2127
|
+
sequence = store->next_sequence;
|
|
2128
|
+
offset = partition->active_write_offset;
|
|
2129
|
+
commit_offset = offset + frame_length - SFS_FRAME_COMMIT_SIZE;
|
|
2130
|
+
memset(partition->active_map + offset, 0, (size_t)frame_length);
|
|
2131
|
+
store_u32_le(partition->active_map + offset, (uint32_t)frame_length);
|
|
2132
|
+
store_u32_le(partition->active_map + offset + 4u, payload_length);
|
|
2133
|
+
store_u64_le(partition->active_map + offset + 8u, sequence);
|
|
2134
|
+
store_u32_le(partition->active_map + offset + 16u,
|
|
2135
|
+
sfs_crc32c((const uint8_t *)payload, payload_length));
|
|
2136
|
+
store_u32_le(partition->active_map + offset + 20u,
|
|
2137
|
+
sfs_crc32c(partition->active_map + offset, 20u));
|
|
2138
|
+
if (payload_length != 0u) {
|
|
2139
|
+
memcpy(partition->active_map + offset + SFS_FRAME_PREFIX_SIZE,
|
|
2140
|
+
payload,
|
|
2141
|
+
payload_length);
|
|
2142
|
+
}
|
|
2143
|
+
atomic_thread_fence(memory_order_release);
|
|
2144
|
+
store_u64_le(partition->active_map + commit_offset,
|
|
2145
|
+
SFS_FRAME_COMMIT_MARKER);
|
|
2146
|
+
atomic_thread_fence(memory_order_release);
|
|
2147
|
+
partition->active_write_offset += frame_length;
|
|
2148
|
+
meta = &partition->segments[partition->segment_count - 1u];
|
|
2149
|
+
meta->write_offset = partition->active_write_offset;
|
|
2150
|
+
if (meta->record_count == 0u) {
|
|
2151
|
+
meta->first_sequence = sequence;
|
|
2152
|
+
}
|
|
2153
|
+
meta->last_sequence = sequence;
|
|
2154
|
+
meta->record_count += 1u;
|
|
2155
|
+
meta->payload_bytes += payload_length;
|
|
2156
|
+
if (partition->record_count == 0u) {
|
|
2157
|
+
partition->first_sequence = sequence;
|
|
2158
|
+
}
|
|
2159
|
+
partition->last_sequence = sequence;
|
|
2160
|
+
partition->record_count += 1u;
|
|
2161
|
+
partition->payload_bytes += payload_length;
|
|
2162
|
+
store->next_sequence += 1u;
|
|
2163
|
+
if (durability == SFS_DURABILITY_SYNC) {
|
|
2164
|
+
result = flush_locked(store, error);
|
|
2165
|
+
if (result != SFS_OK) {
|
|
2166
|
+
goto done;
|
|
2167
|
+
}
|
|
2168
|
+
}
|
|
2169
|
+
out_record->payload_length = payload_length;
|
|
2170
|
+
out_record->partition_id = partition_id;
|
|
2171
|
+
out_record->sequence = sequence;
|
|
2172
|
+
out_record->segment_id = meta->id;
|
|
2173
|
+
out_record->frame_offset = offset;
|
|
2174
|
+
done:
|
|
2175
|
+
(void)pthread_mutex_unlock(&store->mutex);
|
|
2176
|
+
return result;
|
|
2177
|
+
}
|
|
2178
|
+
|
|
2179
|
+
sfs_result_t sfs_scan(sfs_store_t *store,
|
|
2180
|
+
sfs_cursor_t *cursor,
|
|
2181
|
+
void *buffer,
|
|
2182
|
+
uint32_t buffer_capacity,
|
|
2183
|
+
sfs_record_info_t *out_record,
|
|
2184
|
+
sfs_error_t *error)
|
|
2185
|
+
{
|
|
2186
|
+
sfs_result_t result;
|
|
2187
|
+
clear_error(error);
|
|
2188
|
+
if (store == NULL || cursor == NULL || out_record == NULL ||
|
|
2189
|
+
out_record->struct_size < sizeof(*out_record) ||
|
|
2190
|
+
(buffer == NULL && buffer_capacity != 0u) || cursor->flags != 0u ||
|
|
2191
|
+
(cursor->partition_id != SFS_PARTITION_ALL &&
|
|
2192
|
+
cursor->partition_id >= SFS_MAX_PARTITIONS)) {
|
|
2193
|
+
return set_error(error,
|
|
2194
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2195
|
+
0,
|
|
2196
|
+
"invalid scan arguments");
|
|
2197
|
+
}
|
|
2198
|
+
(void)pthread_mutex_lock(&store->mutex);
|
|
2199
|
+
if (cursor->partition_id != SFS_PARTITION_ALL && cursor->segment_id == 0u &&
|
|
2200
|
+
cursor->offset == 0u && cursor->after_sequence != 0u) {
|
|
2201
|
+
result = seek_partition_cursor(store, cursor, buffer, buffer_capacity, out_record, error);
|
|
2202
|
+
} else if (cursor->partition_id != SFS_PARTITION_ALL) {
|
|
2203
|
+
result = scan_partition_cursor(store,
|
|
2204
|
+
cursor,
|
|
2205
|
+
buffer,
|
|
2206
|
+
buffer_capacity,
|
|
2207
|
+
out_record,
|
|
2208
|
+
error);
|
|
2209
|
+
} else {
|
|
2210
|
+
sfs_candidate_t best;
|
|
2211
|
+
uint32_t partition_id;
|
|
2212
|
+
memset(&best, 0, sizeof(best));
|
|
2213
|
+
result = SFS_OK;
|
|
2214
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS;
|
|
2215
|
+
++partition_id) {
|
|
2216
|
+
sfs_candidate_t candidate;
|
|
2217
|
+
if (!store->partitions[partition_id].enabled) {
|
|
2218
|
+
continue;
|
|
2219
|
+
}
|
|
2220
|
+
result = candidate_in_partition(store,
|
|
2221
|
+
partition_id,
|
|
2222
|
+
cursor->after_sequence,
|
|
2223
|
+
&candidate,
|
|
2224
|
+
error);
|
|
2225
|
+
if (result != SFS_OK) {
|
|
2226
|
+
break;
|
|
2227
|
+
}
|
|
2228
|
+
if (candidate.present &&
|
|
2229
|
+
(!best.present ||
|
|
2230
|
+
candidate.frame.sequence < best.frame.sequence)) {
|
|
2231
|
+
best = candidate;
|
|
2232
|
+
}
|
|
2233
|
+
}
|
|
2234
|
+
if (result == SFS_OK && !best.present) {
|
|
2235
|
+
result = SFS_END;
|
|
2236
|
+
} else if (result == SFS_OK) {
|
|
2237
|
+
uint64_t expected = cursor->after_sequence + 1u;
|
|
2238
|
+
result = copy_candidate(store,
|
|
2239
|
+
&best,
|
|
2240
|
+
buffer,
|
|
2241
|
+
buffer_capacity,
|
|
2242
|
+
out_record,
|
|
2243
|
+
error);
|
|
2244
|
+
if (result == SFS_OK) {
|
|
2245
|
+
if (best.frame.sequence > expected) {
|
|
2246
|
+
out_record->flags |= SFS_RECORD_GAP_BEFORE;
|
|
2247
|
+
out_record->gap_first_sequence = expected;
|
|
2248
|
+
out_record->gap_last_sequence = best.frame.sequence - 1u;
|
|
2249
|
+
}
|
|
2250
|
+
cursor->after_sequence = best.frame.sequence;
|
|
2251
|
+
cursor->segment_id = out_record->segment_id;
|
|
2252
|
+
cursor->offset = best.frame_offset + best.frame.total_length;
|
|
2253
|
+
}
|
|
2254
|
+
}
|
|
2255
|
+
}
|
|
2256
|
+
(void)pthread_mutex_unlock(&store->mutex);
|
|
2257
|
+
return result;
|
|
2258
|
+
}
|
|
2259
|
+
|
|
2260
|
+
sfs_result_t sfs_status(sfs_store_t *store,
|
|
2261
|
+
sfs_status_t *out_status,
|
|
2262
|
+
sfs_error_t *error)
|
|
2263
|
+
{
|
|
2264
|
+
uint32_t partition_id;
|
|
2265
|
+
clear_error(error);
|
|
2266
|
+
if (store == NULL || out_status == NULL ||
|
|
2267
|
+
out_status->struct_size < sizeof(*out_status)) {
|
|
2268
|
+
return set_error(error,
|
|
2269
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2270
|
+
0,
|
|
2271
|
+
"invalid status arguments");
|
|
2272
|
+
}
|
|
2273
|
+
(void)pthread_mutex_lock(&store->mutex);
|
|
2274
|
+
memset(out_status, 0, sizeof(*out_status));
|
|
2275
|
+
out_status->struct_size = sizeof(*out_status);
|
|
2276
|
+
out_status->format_version = SFS_FORMAT_VERSION;
|
|
2277
|
+
out_status->flags = store->status_flags;
|
|
2278
|
+
out_status->segment_size = store->segment_size;
|
|
2279
|
+
out_status->next_sequence = store->next_sequence;
|
|
2280
|
+
out_status->recovery_partition_id = store->recovery_partition_id;
|
|
2281
|
+
out_status->recovery_segment_id = store->recovery_segment_id;
|
|
2282
|
+
out_status->recovery_offset = store->recovery_offset;
|
|
2283
|
+
out_status->recovery_discarded_bytes = store->recovery_discarded_bytes;
|
|
2284
|
+
for (partition_id = 0u; partition_id < SFS_MAX_PARTITIONS; ++partition_id) {
|
|
2285
|
+
const sfs_partition_t *source = &store->partitions[partition_id];
|
|
2286
|
+
sfs_partition_status_t *target = &out_status->partitions[partition_id];
|
|
2287
|
+
target->partition_id = partition_id;
|
|
2288
|
+
target->quota_bytes = source->quota_bytes;
|
|
2289
|
+
target->evicted_segments = source->evicted_segments;
|
|
2290
|
+
target->evicted_records = source->evicted_records;
|
|
2291
|
+
target->evicted_payload_bytes = source->evicted_payload_bytes;
|
|
2292
|
+
if (!source->enabled) {
|
|
2293
|
+
continue;
|
|
2294
|
+
}
|
|
2295
|
+
target->flags = SFS_PARTITION_ENABLED;
|
|
2296
|
+
target->allocated_bytes = source->segment_count * store->segment_size;
|
|
2297
|
+
target->segment_count = source->segment_count;
|
|
2298
|
+
target->first_segment_id = source->segment_count == 0u
|
|
2299
|
+
? 0u
|
|
2300
|
+
: source->segments[0].id;
|
|
2301
|
+
target->active_segment_id = source->segment_count == 0u
|
|
2302
|
+
? 0u
|
|
2303
|
+
: source->segments[source->segment_count - 1u]
|
|
2304
|
+
.id;
|
|
2305
|
+
target->active_write_offset = source->active_write_offset;
|
|
2306
|
+
target->record_count = source->record_count;
|
|
2307
|
+
target->payload_bytes = source->payload_bytes;
|
|
2308
|
+
target->first_sequence = source->first_sequence;
|
|
2309
|
+
target->last_sequence = source->last_sequence;
|
|
2310
|
+
out_status->segment_count += source->segment_count;
|
|
2311
|
+
out_status->record_count += source->record_count;
|
|
2312
|
+
out_status->payload_bytes += source->payload_bytes;
|
|
2313
|
+
}
|
|
2314
|
+
(void)pthread_mutex_unlock(&store->mutex);
|
|
2315
|
+
return SFS_OK;
|
|
2316
|
+
}
|
|
2317
|
+
|
|
2318
|
+
sfs_result_t sfs_flush(sfs_store_t *store, sfs_error_t *error)
|
|
2319
|
+
{
|
|
2320
|
+
sfs_result_t result;
|
|
2321
|
+
clear_error(error);
|
|
2322
|
+
if (store == NULL) {
|
|
2323
|
+
return set_error(error,
|
|
2324
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2325
|
+
0,
|
|
2326
|
+
"store is null");
|
|
2327
|
+
}
|
|
2328
|
+
(void)pthread_mutex_lock(&store->mutex);
|
|
2329
|
+
result = flush_locked(store, error);
|
|
2330
|
+
(void)pthread_mutex_unlock(&store->mutex);
|
|
2331
|
+
return result;
|
|
2332
|
+
}
|
|
2333
|
+
|
|
2334
|
+
sfs_result_t sfs_close(sfs_store_t *store, sfs_error_t *error)
|
|
2335
|
+
{
|
|
2336
|
+
sfs_result_t result;
|
|
2337
|
+
clear_error(error);
|
|
2338
|
+
if (store == NULL) {
|
|
2339
|
+
return set_error(error,
|
|
2340
|
+
SFS_ERR_INVALID_ARGUMENT,
|
|
2341
|
+
0,
|
|
2342
|
+
"store is null");
|
|
2343
|
+
}
|
|
2344
|
+
(void)pthread_mutex_lock(&store->mutex);
|
|
2345
|
+
result = flush_locked(store, error);
|
|
2346
|
+
(void)pthread_mutex_unlock(&store->mutex);
|
|
2347
|
+
cleanup_store(store);
|
|
2348
|
+
return result;
|
|
2349
|
+
}
|