@sixb/core 0.0.1 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/README.md +24 -4
  2. package/dist/actions/index.js +17 -17
  3. package/dist/actions/worker.js +7 -7
  4. package/dist/actions/worker.js.map +1 -1
  5. package/dist/agents/adapters.d.ts.map +1 -1
  6. package/dist/agents/builders.d.ts +8 -1
  7. package/dist/agents/builders.d.ts.map +1 -1
  8. package/dist/agents/context.js +2 -2
  9. package/dist/agents/errors.d.ts +15 -0
  10. package/dist/agents/errors.d.ts.map +1 -1
  11. package/dist/agents/index.d.ts +4 -4
  12. package/dist/agents/index.d.ts.map +1 -1
  13. package/dist/agents/index.js +30 -11
  14. package/dist/agents/index.js.map +1 -1
  15. package/dist/agents/tool-definition.d.ts +9 -0
  16. package/dist/agents/tool-definition.d.ts.map +1 -0
  17. package/dist/agents/types.d.ts +64 -0
  18. package/dist/agents/types.d.ts.map +1 -1
  19. package/dist/agents/validation.d.ts +14 -1
  20. package/dist/agents/validation.d.ts.map +1 -1
  21. package/dist/auth/index.js +4 -4
  22. package/dist/authorization/index.js +3 -3
  23. package/dist/blob-storage/browser.js +3 -2
  24. package/dist/blob-storage/browser.js.map +1 -1
  25. package/dist/blob-storage/index.js +4 -3
  26. package/dist/blob-storage/index.js.map +1 -1
  27. package/dist/blob-storage/validation.d.ts.map +1 -1
  28. package/dist/bootstrap/index.js +25 -25
  29. package/dist/broker/index.js +2 -2
  30. package/dist/{chunk-590dejbp.js → chunk-4hwxq5e6.js} +1 -1
  31. package/dist/{chunk-wj01s5hr.js → chunk-4tvj3ate.js} +1 -1
  32. package/dist/{chunk-tsk2bxs3.js → chunk-53s4asfm.js} +13 -13
  33. package/dist/{chunk-tsk2bxs3.js.map → chunk-53s4asfm.js.map} +3 -3
  34. package/dist/{chunk-8r8cw2xw.js → chunk-63te770v.js} +40 -5
  35. package/dist/{chunk-8r8cw2xw.js.map → chunk-63te770v.js.map} +3 -3
  36. package/dist/{chunk-vj723r3p.js → chunk-6rzdtvn5.js} +1 -1
  37. package/dist/{chunk-5yny9z0e.js → chunk-6zrpewy4.js} +1 -1
  38. package/dist/{chunk-nvamz5dv.js → chunk-86459jn0.js} +294 -22
  39. package/dist/chunk-86459jn0.js.map +22 -0
  40. package/dist/{chunk-p6r5g77m.js → chunk-8kv2sez0.js} +1 -1
  41. package/dist/{chunk-k41ajtz7.js → chunk-9k3hgv8h.js} +1 -1
  42. package/dist/{chunk-jdwx7vzs.js → chunk-avppcgt6.js} +7 -7
  43. package/dist/{chunk-jdwx7vzs.js.map → chunk-avppcgt6.js.map} +1 -1
  44. package/dist/{chunk-2gpr6jsz.js → chunk-b1bqvhpe.js} +2 -2
  45. package/dist/{chunk-pv1yad91.js → chunk-b5dp7gaf.js} +17 -90
  46. package/dist/{chunk-pv1yad91.js.map → chunk-b5dp7gaf.js.map} +5 -6
  47. package/dist/{chunk-1vffvydj.js → chunk-be9vm29v.js} +6 -6
  48. package/dist/chunk-be9vm29v.js.map +10 -0
  49. package/dist/{chunk-wxhdnwh4.js → chunk-bh8bwma3.js} +1 -1
  50. package/dist/{chunk-0psb6kdw.js → chunk-btcsywk4.js} +12 -16
  51. package/dist/{chunk-0psb6kdw.js.map → chunk-btcsywk4.js.map} +3 -3
  52. package/dist/{chunk-0fygrz37.js → chunk-exrfbt2d.js} +1 -1
  53. package/dist/{chunk-c3zd880g.js → chunk-f6kyyeef.js} +2 -2
  54. package/dist/{chunk-q4qzq0ea.js → chunk-hb9xcx9c.js} +3 -3
  55. package/dist/{chunk-4pycxcj3.js → chunk-hz8m927w.js} +6 -9
  56. package/dist/{chunk-4pycxcj3.js.map → chunk-hz8m927w.js.map} +3 -3
  57. package/dist/{chunk-btpe64e4.js → chunk-m9w0artv.js} +1 -1
  58. package/dist/{chunk-45tw90k7.js → chunk-md9jbrm9.js} +1 -1
  59. package/dist/{chunk-dhmj53mx.js → chunk-nbywkytr.js} +3 -3
  60. package/dist/{chunk-t358gywy.js → chunk-p7vj8xgb.js} +8 -8
  61. package/dist/{chunk-2s6sydsh.js → chunk-qzyemcj7.js} +294 -4
  62. package/dist/chunk-qzyemcj7.js.map +13 -0
  63. package/dist/{chunk-r65yscvk.js → chunk-r82k4z7z.js} +131 -13
  64. package/dist/chunk-r82k4z7z.js.map +13 -0
  65. package/dist/{chunk-aebzcsxr.js → chunk-raq0x6bd.js} +11 -9
  66. package/dist/{chunk-aebzcsxr.js.map → chunk-raq0x6bd.js.map} +4 -4
  67. package/dist/{chunk-hsv43v58.js → chunk-rgzed50y.js} +77 -22
  68. package/dist/chunk-rgzed50y.js.map +22 -0
  69. package/dist/{chunk-yvef9xwc.js → chunk-rjqm6ksg.js} +7 -7
  70. package/dist/{chunk-f2b5e2gf.js → chunk-sb0shdvv.js} +18 -3
  71. package/dist/{chunk-f2b5e2gf.js.map → chunk-sb0shdvv.js.map} +3 -3
  72. package/dist/{chunk-dcgrzj1t.js → chunk-xt6qx9sh.js} +1 -1
  73. package/dist/{chunk-mx84y05q.js → chunk-y2ygpd0m.js} +2 -2
  74. package/dist/{chunk-yxss87qa.js → chunk-yn947ck3.js} +1 -1
  75. package/dist/datasets/builders.d.ts +18 -5
  76. package/dist/datasets/builders.d.ts.map +1 -1
  77. package/dist/datasets/changes.d.ts +18 -0
  78. package/dist/datasets/changes.d.ts.map +1 -0
  79. package/dist/datasets/index.d.ts +3 -1
  80. package/dist/datasets/index.d.ts.map +1 -1
  81. package/dist/datasets/types.d.ts +2 -0
  82. package/dist/datasets/types.d.ts.map +1 -1
  83. package/dist/datasets/validation.d.ts.map +1 -1
  84. package/dist/events/index.js +2 -2
  85. package/dist/http/api-routes.d.ts.map +1 -1
  86. package/dist/http/index.js +1 -1
  87. package/dist/index.d.ts +6 -6
  88. package/dist/index.d.ts.map +1 -1
  89. package/dist/index.js +64 -58
  90. package/dist/index.js.map +1 -1
  91. package/dist/json.d.ts +8 -1
  92. package/dist/json.d.ts.map +1 -1
  93. package/dist/lake-storage/definition-updates.d.ts.map +1 -1
  94. package/dist/lake-storage/in-memory.d.ts +15 -1
  95. package/dist/lake-storage/in-memory.d.ts.map +1 -1
  96. package/dist/lake-storage/index.d.ts +3 -1
  97. package/dist/lake-storage/index.d.ts.map +1 -1
  98. package/dist/lake-storage/index.js +15 -7
  99. package/dist/lake-storage/index.js.map +1 -1
  100. package/dist/lake-storage/merge-validation.d.ts +10 -0
  101. package/dist/lake-storage/merge-validation.d.ts.map +1 -0
  102. package/dist/lake-storage/merge.d.ts +32 -0
  103. package/dist/lake-storage/merge.d.ts.map +1 -0
  104. package/dist/lake-storage/types.d.ts +3 -1
  105. package/dist/lake-storage/types.d.ts.map +1 -1
  106. package/dist/logging/index.js +3 -3
  107. package/dist/logging/types.js +2 -2
  108. package/dist/materialization/index.js +2 -2
  109. package/dist/materializer/index.js +15 -15
  110. package/dist/materializer/index.js.map +1 -1
  111. package/dist/objects/index.js +17 -17
  112. package/dist/objects/query/index.js +8 -8
  113. package/dist/ontology/index.js +5 -5
  114. package/dist/ontology/internal.js +7 -6
  115. package/dist/ontology/internal.js.map +3 -3
  116. package/dist/ontology/validation/normalize.d.ts.map +1 -1
  117. package/dist/projections/internal.js +4 -4
  118. package/dist/runtime/sixb.d.ts.map +1 -1
  119. package/dist/storage/index.d.ts +1 -1
  120. package/dist/storage/index.d.ts.map +1 -1
  121. package/dist/storage/index.js +16 -16
  122. package/dist/storage/index.js.map +1 -1
  123. package/dist/storage/ontology/provider.js +4 -4
  124. package/dist/storage/sync-runs/index.d.ts +1 -1
  125. package/dist/storage/sync-runs/index.d.ts.map +1 -1
  126. package/dist/storage/sync-runs/types.d.ts +7 -3
  127. package/dist/storage/sync-runs/types.d.ts.map +1 -1
  128. package/dist/storage/types.d.ts +1 -1
  129. package/dist/storage/types.d.ts.map +1 -1
  130. package/dist/storage/workflow-runs/in-memory.d.ts.map +1 -1
  131. package/dist/storage/workflow-runs/types.d.ts +2 -0
  132. package/dist/storage/workflow-runs/types.d.ts.map +1 -1
  133. package/dist/syncs/builders.d.ts +11 -3
  134. package/dist/syncs/builders.d.ts.map +1 -1
  135. package/dist/syncs/index.d.ts +1 -1
  136. package/dist/syncs/index.d.ts.map +1 -1
  137. package/dist/syncs/types.d.ts +24 -19
  138. package/dist/syncs/types.d.ts.map +1 -1
  139. package/dist/syncs/validation.d.ts +14 -0
  140. package/dist/syncs/validation.d.ts.map +1 -0
  141. package/dist/testing/index.d.ts +1 -0
  142. package/dist/testing/index.d.ts.map +1 -1
  143. package/dist/testing/index.js +837 -583
  144. package/dist/testing/index.js.map +5 -4
  145. package/dist/testing/lake-merge-storage-contract.d.ts +7 -0
  146. package/dist/testing/lake-merge-storage-contract.d.ts.map +1 -0
  147. package/dist/testing/lake-storage-contract.d.ts.map +1 -1
  148. package/dist/workflows/index.js +23 -23
  149. package/package.json +1 -1
  150. package/src/agents/adapters.ts +1 -9
  151. package/src/agents/builders.ts +53 -3
  152. package/src/agents/errors.ts +22 -0
  153. package/src/agents/index.ts +22 -2
  154. package/src/agents/prompt.ts +2 -2
  155. package/src/agents/tool-definition.ts +93 -0
  156. package/src/agents/types.ts +101 -0
  157. package/src/agents/validation.ts +287 -3
  158. package/src/blob-storage/validation.ts +2 -10
  159. package/src/datasets/builders.ts +49 -5
  160. package/src/datasets/changes.ts +13 -0
  161. package/src/datasets/index.ts +3 -0
  162. package/src/datasets/types.ts +5 -0
  163. package/src/datasets/validation.ts +53 -11
  164. package/src/http/api-routes.ts +39 -4
  165. package/src/index.ts +18 -1
  166. package/src/json.ts +11 -4
  167. package/src/lake-storage/definition-updates.ts +21 -0
  168. package/src/lake-storage/in-memory.ts +284 -2
  169. package/src/lake-storage/index.ts +14 -0
  170. package/src/lake-storage/merge-validation.ts +117 -0
  171. package/src/lake-storage/merge.ts +35 -0
  172. package/src/lake-storage/types.ts +3 -1
  173. package/src/ontology/json-schema.ts +2 -1
  174. package/src/ontology/validation/normalize.ts +9 -15
  175. package/src/runtime/sixb.ts +16 -1
  176. package/src/storage/index.ts +1 -0
  177. package/src/storage/sync-runs/in-memory.ts +2 -2
  178. package/src/storage/sync-runs/index.ts +1 -0
  179. package/src/storage/sync-runs/types.ts +9 -4
  180. package/src/storage/types.ts +1 -0
  181. package/src/storage/workflow-runs/in-memory.ts +2 -0
  182. package/src/storage/workflow-runs/types.ts +2 -0
  183. package/src/syncs/builders.ts +31 -13
  184. package/src/syncs/index.ts +1 -0
  185. package/src/syncs/types.ts +44 -18
  186. package/src/syncs/validation.ts +77 -0
  187. package/src/testing/index.ts +4 -0
  188. package/src/testing/lake-merge-storage-contract.ts +302 -0
  189. package/src/testing/lake-storage-contract.ts +23 -0
  190. package/src/workflows/snapshot.ts +1 -10
  191. package/dist/chunk-1vffvydj.js.map +0 -10
  192. package/dist/chunk-2s6sydsh.js.map +0 -12
  193. package/dist/chunk-hsv43v58.js.map +0 -21
  194. package/dist/chunk-nvamz5dv.js.map +0 -21
  195. package/dist/chunk-r65yscvk.js.map +0 -11
  196. /package/dist/{chunk-590dejbp.js.map → chunk-4hwxq5e6.js.map} +0 -0
  197. /package/dist/{chunk-wj01s5hr.js.map → chunk-4tvj3ate.js.map} +0 -0
  198. /package/dist/{chunk-vj723r3p.js.map → chunk-6rzdtvn5.js.map} +0 -0
  199. /package/dist/{chunk-5yny9z0e.js.map → chunk-6zrpewy4.js.map} +0 -0
  200. /package/dist/{chunk-p6r5g77m.js.map → chunk-8kv2sez0.js.map} +0 -0
  201. /package/dist/{chunk-k41ajtz7.js.map → chunk-9k3hgv8h.js.map} +0 -0
  202. /package/dist/{chunk-2gpr6jsz.js.map → chunk-b1bqvhpe.js.map} +0 -0
  203. /package/dist/{chunk-wxhdnwh4.js.map → chunk-bh8bwma3.js.map} +0 -0
  204. /package/dist/{chunk-0fygrz37.js.map → chunk-exrfbt2d.js.map} +0 -0
  205. /package/dist/{chunk-c3zd880g.js.map → chunk-f6kyyeef.js.map} +0 -0
  206. /package/dist/{chunk-q4qzq0ea.js.map → chunk-hb9xcx9c.js.map} +0 -0
  207. /package/dist/{chunk-btpe64e4.js.map → chunk-m9w0artv.js.map} +0 -0
  208. /package/dist/{chunk-45tw90k7.js.map → chunk-md9jbrm9.js.map} +0 -0
  209. /package/dist/{chunk-dhmj53mx.js.map → chunk-nbywkytr.js.map} +0 -0
  210. /package/dist/{chunk-t358gywy.js.map → chunk-p7vj8xgb.js.map} +0 -0
  211. /package/dist/{chunk-yvef9xwc.js.map → chunk-rjqm6ksg.js.map} +0 -0
  212. /package/dist/{chunk-dcgrzj1t.js.map → chunk-xt6qx9sh.js.map} +0 -0
  213. /package/dist/{chunk-mx84y05q.js.map → chunk-y2ygpd0m.js.map} +0 -0
  214. /package/dist/{chunk-yxss87qa.js.map → chunk-yn947ck3.js.map} +0 -0
@@ -0,0 +1,117 @@
1
+ import type { DatasetDefinition, MergeChange } from "../datasets"
2
+ import { getDatasetRowValidationError } from "../datasets"
3
+ import { isPlainRecord } from "../json"
4
+ import { LakeStorageError } from "./errors"
5
+ import type { DatasetRow } from "./types"
6
+
7
+ /** Return a new ordered primary-key column list, or `null` for an unkeyed dataset. */
8
+ export function getDatasetPrimaryKeyColumns(dataset: DatasetDefinition): readonly string[] | null {
9
+ if (dataset.primaryKey === undefined) {
10
+ return null
11
+ }
12
+ return typeof dataset.primaryKey === "string" ? [dataset.primaryKey] : [...dataset.primaryKey]
13
+ }
14
+
15
+ /** Collision-safe internal identity for a validated row or delete-key object. */
16
+ export function encodeDatasetPrimaryKey(dataset: DatasetDefinition, value: DatasetRow): string {
17
+ const columns = getDatasetPrimaryKeyColumns(dataset)
18
+ if (columns === null) {
19
+ throw new LakeStorageError(
20
+ `[SixbLake] Dataset '${dataset.id}' must define a primaryKey before rows can be keyed.`
21
+ )
22
+ }
23
+
24
+ const values = columns.map((columnName) => {
25
+ const columnValue = value[columnName]
26
+ if (typeof columnValue !== "string") {
27
+ throw new LakeStorageError(
28
+ `[SixbLake] Dataset '${dataset.id}' primary-key column '${columnName}' must be a string.`
29
+ )
30
+ }
31
+ return columnValue
32
+ })
33
+ return JSON.stringify(values)
34
+ }
35
+
36
+ export function getDatasetMergeChangeValidationError(
37
+ change: unknown,
38
+ dataset: DatasetDefinition
39
+ ): string | null {
40
+ const primaryKeyColumns = getDatasetPrimaryKeyColumns(dataset)
41
+ if (primaryKeyColumns === null) {
42
+ return `Dataset '${dataset.id}' must define a primaryKey before it can be merged.`
43
+ }
44
+ if (!isPlainRecord(change)) {
45
+ return `Dataset '${dataset.id}' merge changes must be plain objects.`
46
+ }
47
+
48
+ if (change.kind === "upsert") {
49
+ const shapeError = getChangeShapeError(change, dataset, "row")
50
+ if (shapeError) {
51
+ return shapeError
52
+ }
53
+ return getDatasetRowValidationError(change.row, dataset)
54
+ }
55
+
56
+ if (change.kind === "delete") {
57
+ const shapeError = getChangeShapeError(change, dataset, "key")
58
+ if (shapeError) {
59
+ return shapeError
60
+ }
61
+ return getDeleteKeyValidationError(change.key, dataset, primaryKeyColumns)
62
+ }
63
+
64
+ return `Dataset '${dataset.id}' merge change kind must be 'upsert' or 'delete'.`
65
+ }
66
+
67
+ /** Clone a validated change before a provider stages it beyond the caller's ownership. */
68
+ export function cloneDatasetMergeChange(
69
+ change: MergeChange<DatasetRow, DatasetRow>
70
+ ): MergeChange<DatasetRow, DatasetRow> {
71
+ return change.kind === "upsert"
72
+ ? { kind: "upsert", row: structuredClone(change.row) }
73
+ : { kind: "delete", key: structuredClone(change.key) }
74
+ }
75
+
76
+ function getChangeShapeError(
77
+ change: Record<string, unknown>,
78
+ dataset: DatasetDefinition,
79
+ payloadField: "row" | "key"
80
+ ): string | null {
81
+ const allowedFields = new Set(["kind", payloadField])
82
+ for (const field of Object.keys(change)) {
83
+ if (!allowedFields.has(field)) {
84
+ return `Dataset '${dataset.id}' ${change.kind as string} change contains unknown field '${field}'.`
85
+ }
86
+ }
87
+ if (!Object.hasOwn(change, payloadField)) {
88
+ return `Dataset '${dataset.id}' ${change.kind as string} change is missing '${payloadField}'.`
89
+ }
90
+ return null
91
+ }
92
+
93
+ function getDeleteKeyValidationError(
94
+ key: unknown,
95
+ dataset: DatasetDefinition,
96
+ primaryKeyColumns: readonly string[]
97
+ ): string | null {
98
+ if (!isPlainRecord(key)) {
99
+ return `Dataset '${dataset.id}' delete keys must be plain objects.`
100
+ }
101
+
102
+ const expectedColumns = new Set(primaryKeyColumns)
103
+ for (const columnName of Object.keys(key)) {
104
+ if (!expectedColumns.has(columnName)) {
105
+ return `Dataset '${dataset.id}' delete key contains unknown column '${columnName}'.`
106
+ }
107
+ }
108
+ for (const columnName of primaryKeyColumns) {
109
+ if (!Object.hasOwn(key, columnName)) {
110
+ return `Dataset '${dataset.id}' delete key is missing primary-key column '${columnName}'.`
111
+ }
112
+ if (typeof key[columnName] !== "string") {
113
+ return `Dataset '${dataset.id}' delete key column '${columnName}' must be a string.`
114
+ }
115
+ }
116
+ return null
117
+ }
@@ -0,0 +1,35 @@
1
+ import type { DatasetDefinition, MergeChange } from "../datasets"
2
+ import type { DatasetProducer, DatasetRow, DatasetVersion, DatasetVersionRef } from "./types"
3
+
4
+ export interface BeginDatasetMergeInput {
5
+ /** Stored definition for the keyed dataset whose latest version is captured by `beginMerge`. */
6
+ readonly dataset: DatasetDefinition
7
+ /** Optional caller guard checked against the same latest version captured by the session. */
8
+ readonly expectedLatestVersionId?: string
9
+ readonly producer?: DatasetProducer
10
+ readonly inputs?: readonly DatasetVersionRef[]
11
+ }
12
+
13
+ export interface CommitDatasetMergeInput {
14
+ readonly commitMessage?: string
15
+ }
16
+
17
+ /**
18
+ * A merge can be unchanged before the dataset has a first version, so its result carries the
19
+ * version separately instead of extending `DatasetVersion` like ordinary write commits do.
20
+ */
21
+ export type DatasetMergeCommitResult =
22
+ | { readonly outcome: "created"; readonly version: DatasetVersion }
23
+ | { readonly outcome: "unchanged"; readonly version: DatasetVersion | null }
24
+
25
+ export interface LakeMergeSession {
26
+ /** Stage ordered complete-row upserts and exact primary-key deletes. */
27
+ writeChanges(
28
+ changes:
29
+ | Iterable<MergeChange<DatasetRow, DatasetRow>>
30
+ | AsyncIterable<MergeChange<DatasetRow, DatasetRow>>
31
+ ): Promise<void>
32
+ /** Commit only if the latest version still matches the version captured when the session began. */
33
+ commit(input?: CommitDatasetMergeInput): Promise<DatasetMergeCommitResult>
34
+ abort(): Promise<void>
35
+ }
@@ -1,7 +1,8 @@
1
1
  import type { DatasetDefinition, DatasetSchema } from "../datasets"
2
+ import type { BeginDatasetMergeInput, LakeMergeSession } from "./merge"
2
3
 
3
4
  export type DatasetWriteMode = "snapshot" | "append"
4
- export type DatasetVersionMode = DatasetWriteMode | "schema"
5
+ export type DatasetVersionMode = DatasetWriteMode | "merge" | "schema"
5
6
 
6
7
  export type DatasetRow = Readonly<Record<string, unknown>>
7
8
 
@@ -124,6 +125,7 @@ export interface LakeStorage {
124
125
  listVersions(datasetId: string, limit?: number): Promise<readonly DatasetVersion[]>
125
126
 
126
127
  beginWrite(input: BeginDatasetWriteInput): Promise<LakeWriteSession>
128
+ beginMerge(input: BeginDatasetMergeInput): Promise<LakeMergeSession>
127
129
 
128
130
  getLatestVersion(datasetId: string): Promise<DatasetVersion | null>
129
131
  getVersion(datasetId: string, versionId: string): Promise<DatasetVersion | null>
@@ -53,8 +53,9 @@ function schemaJsonSchema(
53
53
  case "integer":
54
54
  return { type: "integer" }
55
55
  case "double":
56
- case "decimal":
57
56
  return { type: "number" }
57
+ case "decimal":
58
+ return { type: "string", pattern: "^[+-]?\\d+(?:\\.\\d+)?$" }
58
59
  case "boolean":
59
60
  return { type: "boolean" }
60
61
  case "date":
@@ -2,7 +2,7 @@ import { assertJsonValue, cloneJsonValue, type JsonValue } from "../../json"
2
2
  import type { ObjectFieldSchema, Property, Schema, ValueType } from ".."
3
3
  import { normalizeDecimalValue } from "../decimal"
4
4
  import { OntologyValidationError } from "../errors"
5
- import { resolveValueTypeSchema } from "./schema"
5
+ import { isRecord, resolveValueTypeSchema } from "./schema"
6
6
 
7
7
  /**
8
8
  * Normalize a property bag to JSON-safe values using each property's schema —
@@ -84,20 +84,18 @@ export function normalizeSchemaValue(
84
84
 
85
85
  const fields = schema.properties
86
86
  const record = assertPlainRecord(value, path)
87
- const output: Record<string, JsonValue> = {}
87
+ const entries: Array<[string, JsonValue]> = []
88
88
  for (const [fieldId, field] of Object.entries(fields)) {
89
89
  const fieldValue = record[fieldId]
90
90
  if (fieldValue === undefined) {
91
91
  continue
92
92
  }
93
- output[fieldId] = normalizeObjectFieldValue(
94
- field,
95
- fieldValue,
96
- `${path}.${fieldId}`,
97
- valueTypesById
98
- )
93
+ entries.push([
94
+ fieldId,
95
+ normalizeObjectFieldValue(field, fieldValue, `${path}.${fieldId}`, valueTypesById),
96
+ ])
99
97
  }
100
- return output
98
+ return Object.fromEntries(entries)
101
99
  }
102
100
 
103
101
  function normalizeObjectFieldValue(
@@ -149,10 +147,6 @@ function normalizeDateLike(value: unknown, path: string): Date {
149
147
  return date
150
148
  }
151
149
 
152
- function isPlainRecord(value: unknown): value is Record<string, unknown> {
153
- return typeof value === "object" && value !== null && !Array.isArray(value)
154
- }
155
-
156
150
  /**
157
151
  * Inverse of `normalizeSchemaValue` for the handler-facing surface: re-hydrate
158
152
  * `date`/`timestamp` values from their stored ISO string back into a `Date`, so
@@ -196,7 +190,7 @@ export function coerceSchemaValueToTyped(
196
190
  }
197
191
 
198
192
  if (schema.type === "map") {
199
- if (!isPlainRecord(value)) return value
193
+ if (!isRecord(value)) return value
200
194
  return Object.fromEntries(
201
195
  Object.entries(value).map(([key, entry]) => [
202
196
  key,
@@ -205,7 +199,7 @@ export function coerceSchemaValueToTyped(
205
199
  )
206
200
  }
207
201
 
208
- if (!isPlainRecord(value)) return value
202
+ if (!isRecord(value)) return value
209
203
  const output: Record<string, unknown> = { ...value }
210
204
  for (const [fieldId, field] of Object.entries(schema.properties)) {
211
205
  if (value[fieldId] === undefined) continue
@@ -19,7 +19,7 @@ import {
19
19
  } from "../actions/request"
20
20
  import type { ActionDefinition } from "../actions/types"
21
21
  import type { AgentDefinition, RequestAgentRunInput, RequestAgentRunResult } from "../agents"
22
- import { AgentsRuntime, validateAgentGroupReferences } from "../agents"
22
+ import { AgentsRuntime, validateAgentGroupReferences, validateAgentToolsAtStartup } from "../agents"
23
23
  import {
24
24
  AuthRuntime,
25
25
  AuthRuntimeError,
@@ -96,6 +96,10 @@ import {
96
96
  requestSyncRun,
97
97
  type SyncRunRequestResult,
98
98
  } from "../syncs/request"
99
+ import {
100
+ validateKeyedDatasetWriterTopology,
101
+ validateMergeSyncProjectionSafety,
102
+ } from "../syncs/validation"
99
103
  import type { RegisteredWebhook } from "../webhooks"
100
104
  import { registerWebhooks, WebhookValidationError, webhookRoute } from "../webhooks"
101
105
  import type {
@@ -227,6 +231,7 @@ export class Sixb<TOntologySources extends readonly OntologySource[]>
227
231
  this.actionRegistry = new ActionRegistry(options.actions ?? [], this.ontology)
228
232
  const registeredActionIds = new Set(this.actionRegistry.list().map((action) => action.id))
229
233
  const agents = options.agents ?? []
234
+ validateAgentToolsAtStartup(agents)
230
235
  this.security = createRuntimeSecurityRegistry({
231
236
  groups: options.groups ?? [],
232
237
  roles: options.roles ?? [],
@@ -327,6 +332,12 @@ export class Sixb<TOntologySources extends readonly OntologySource[]>
327
332
  this.pipelinesById.set(pipeline.id, pipeline)
328
333
  }
329
334
 
335
+ validateKeyedDatasetWriterTopology({
336
+ datasetsById: this.datasetsById,
337
+ syncs: [...this.syncsById.values()],
338
+ pipelines: [...this.pipelinesById.values()],
339
+ })
340
+
330
341
  // Rules validate against the resolved ontology so inherited properties and
331
342
  // links are available before checking predicate references.
332
343
  validateRulesAtStartup(this.rules, this.ontology)
@@ -368,6 +379,10 @@ export class Sixb<TOntologySources extends readonly OntologySource[]>
368
379
  ontology: this.ontology,
369
380
  datasetsById: this.datasetsById,
370
381
  })
382
+ validateMergeSyncProjectionSafety({
383
+ syncs: [...this.syncsById.values()],
384
+ telemetryProjections: this.projectionRegistry.listTelemetryProjections(),
385
+ })
371
386
  registerProjectionRegistry(this, this.projectionRegistry)
372
387
 
373
388
  const materializer = createOntologyMaterializer({
@@ -388,6 +388,7 @@ export type {
388
388
  ListSyncRunsResult,
389
389
  StartSyncRunInput,
390
390
  SyncRunFailure,
391
+ SyncRunMode,
391
392
  SyncRunRecord,
392
393
  SyncRunStatus,
393
394
  SyncRunStorage,
@@ -90,9 +90,9 @@ export class InMemorySyncRunStorage implements SyncRunStorage {
90
90
  `[Sixb] Sync run '${input.id}' output dataset '${input.output.datasetId}' does not match '${existing.datasetId}'.`
91
91
  )
92
92
  }
93
- if (!input.output && (existing.mode !== "append" || input.rowsRead !== 0)) {
93
+ if (!input.output && input.rowsRead !== 0 && existing.mode !== "merge") {
94
94
  throw new SyncRunError(
95
- `[Sixb] Sync run '${input.id}' may omit its output only for an empty append.`
95
+ `[Sixb] Sync run '${input.id}' may omit its output with rows read only for an initial merge no-op.`
96
96
  )
97
97
  }
98
98
  }
@@ -8,6 +8,7 @@ export type {
8
8
  ListSyncRunsResult,
9
9
  StartSyncRunInput,
10
10
  SyncRunFailure,
11
+ SyncRunMode,
11
12
  SyncRunRecord,
12
13
  SyncRunStatus,
13
14
  SyncRunStorage,
@@ -1,6 +1,8 @@
1
1
  import type { JsonValue } from "../../json"
2
2
  import type { DatasetVersionRef, DatasetWriteMode } from "../../lake-storage"
3
3
 
4
+ export type SyncRunMode = DatasetWriteMode | "merge"
5
+
4
6
  export type SyncRunStatus = "running" | "succeeded" | "failed" | "cancelled"
5
7
 
6
8
  export interface SyncRunFailure {
@@ -13,11 +15,12 @@ export interface SyncRunRecord {
13
15
  readonly projectId: string
14
16
  readonly syncId: string
15
17
  readonly datasetId: string
16
- readonly mode: DatasetWriteMode
18
+ readonly mode: SyncRunMode
17
19
  readonly status: SyncRunStatus
18
20
  readonly startedAt: Date
19
21
  readonly finishedAt?: Date
20
- readonly rowsRead?: number // Rows produced by this run, not the dataset's full visible row count.
22
+ /** Source items successfully consumed. Merge runs count both upserts and deletes. */
23
+ readonly rowsRead?: number
21
24
  readonly output?: DatasetVersionRef
22
25
  readonly expectedLatestVersionId?: string
23
26
  readonly commitMessage?: string
@@ -30,7 +33,7 @@ export interface StartSyncRunInput {
30
33
  readonly projectId: string
31
34
  readonly syncId: string
32
35
  readonly datasetId: string
33
- readonly mode: DatasetWriteMode
36
+ readonly mode: SyncRunMode
34
37
  readonly startedAt?: Date
35
38
  readonly expectedLatestVersionId?: string
36
39
  readonly commitMessage?: string
@@ -42,8 +45,9 @@ export type FinishSyncRunInput =
42
45
  readonly projectId: string
43
46
  readonly status: "succeeded"
44
47
  readonly finishedAt?: Date
48
+ /** Source items successfully consumed. Merge runs count both upserts and deletes. */
45
49
  readonly rowsRead: number
46
- /** Absent only when the first append run succeeds without producing any rows. */
50
+ /** Absent when a successful run has no dataset version to reference. */
47
51
  readonly output?: DatasetVersionRef
48
52
  readonly checkpoint?: JsonValue
49
53
  }
@@ -52,6 +56,7 @@ export type FinishSyncRunInput =
52
56
  readonly projectId: string
53
57
  readonly status: "failed" | "cancelled"
54
58
  readonly finishedAt?: Date
59
+ /** Source items successfully consumed before the failure or cancellation. */
55
60
  readonly rowsRead?: number
56
61
  readonly error?: SyncRunFailure
57
62
  }
@@ -172,6 +172,7 @@ export type {
172
172
  ListSyncRunsResult,
173
173
  StartSyncRunInput,
174
174
  SyncRunFailure,
175
+ SyncRunMode,
175
176
  SyncRunRecord,
176
177
  SyncRunStatus,
177
178
  SyncRunStorage,
@@ -174,10 +174,12 @@ export class InMemoryWorkflowRunStorage implements WorkflowRunStorage {
174
174
  input.status === "succeeded"
175
175
  ? {
176
176
  ...base,
177
+ output: cloneRecord(input.output),
177
178
  error: undefined,
178
179
  }
179
180
  : {
180
181
  ...base,
182
+ output: undefined,
181
183
  error: input.error,
182
184
  }
183
185
 
@@ -121,6 +121,7 @@ export interface WorkflowRunRecord {
121
121
  readonly workflowId: string
122
122
  readonly status: WorkflowRunStatus
123
123
  readonly input: WorkflowIOSnapshot
124
+ readonly output?: WorkflowIOSnapshot
124
125
  readonly queuedAt?: Date
125
126
  readonly startedAt: Date
126
127
  readonly finishedAt?: Date
@@ -201,6 +202,7 @@ export type FinishWorkflowRunInput =
201
202
  readonly id: string
202
203
  readonly projectId: string
203
204
  readonly status: "succeeded"
205
+ readonly output: WorkflowIOSnapshot
204
206
  readonly finishedAt?: Date
205
207
  readonly executionToken?: string
206
208
  }
@@ -8,6 +8,7 @@ import type {
8
8
  BatchSyncDefinitionConfig,
9
9
  SyncBuilder,
10
10
  SyncDefinition,
11
+ SyncMode,
11
12
  SyncReadBuilder,
12
13
  SyncTargetBuilder,
13
14
  } from "./types"
@@ -18,16 +19,19 @@ function assertNonEmpty(value: string, field: string): void {
18
19
  }
19
20
  }
20
21
 
21
- function assertDataset(dataset: DatasetDefinition): void {
22
+ function assertDataset(dataset: DatasetDefinition, mode: SyncMode): void {
22
23
  assertNonEmpty(dataset.id, "dataset id")
24
+ if (mode === "merge" && dataset.primaryKey === undefined) {
25
+ throw new SyncValidationError(`Merge sync dataset '${dataset.id}' must define a primaryKey.`)
26
+ }
23
27
  }
24
28
 
25
29
  function normalizeBatchSyncConfig(options: BatchSyncConfig | undefined): BatchSyncDefinitionConfig {
26
30
  const mode = options?.mode ?? "snapshot"
27
31
 
28
- if (mode !== "snapshot" && mode !== "append") {
32
+ if (mode !== "snapshot" && mode !== "append" && mode !== "merge") {
29
33
  throw new SyncValidationError(
30
- `Invalid sync mode '${String(mode)}'. Expected 'snapshot' or 'append'.`
34
+ `Invalid sync mode '${String(mode)}'. Expected 'snapshot', 'append', or 'merge'.`
31
35
  )
32
36
  }
33
37
 
@@ -37,43 +41,57 @@ function normalizeBatchSyncConfig(options: BatchSyncConfig | undefined): BatchSy
37
41
  }
38
42
  }
39
43
 
44
+ type DefaultBatchSyncConfig = Omit<BatchSyncConfig, "mode"> & { readonly mode?: undefined }
45
+
40
46
  /**
41
47
  * Define a batch sync that reads from one connector and writes into one dataset.
42
48
  *
43
49
  * The returned definition is inert and safe to export from `syncs/` modules.
44
- * V1 supports `snapshot` and `append` modes with optional triggers and checkpoints.
50
+ * Supports `snapshot`, `append`, and keyed `merge` modes with optional triggers and checkpoints.
45
51
  */
52
+ export function defineSync<TId extends string>(
53
+ id: TId,
54
+ options?: DefaultBatchSyncConfig
55
+ ): SyncBuilder<TId, never, "snapshot">
56
+ export function defineSync<TId extends string, const TMode extends SyncMode>(
57
+ id: TId,
58
+ options: BatchSyncConfig<TMode> & { readonly mode: TMode }
59
+ ): SyncBuilder<TId, never, TMode>
60
+ export function defineSync<TId extends string>(
61
+ id: TId,
62
+ options: BatchSyncConfig
63
+ ): SyncBuilder<TId, never, SyncMode>
46
64
  export function defineSync<TId extends string>(
47
65
  id: TId,
48
66
  options?: BatchSyncConfig
49
- ): SyncBuilder<TId> {
67
+ ): SyncBuilder<TId, never, SyncMode> {
50
68
  assertNonEmpty(id, "id")
51
69
 
52
70
  const config = normalizeBatchSyncConfig(options)
53
71
  const triggers: ScheduleReference[] = []
54
72
 
55
- function createBuilder<TCheckpoint>(): SyncBuilder<TId, TCheckpoint> {
56
- const builder: SyncBuilder<TId, TCheckpoint> = {
57
- when(schedule: ScheduleDefinition): SyncBuilder<TId, TCheckpoint> {
73
+ function createBuilder<TCheckpoint>(): SyncBuilder<TId, TCheckpoint, SyncMode> {
74
+ const builder: SyncBuilder<TId, TCheckpoint, SyncMode> = {
75
+ when(schedule: ScheduleDefinition): SyncBuilder<TId, TCheckpoint, SyncMode> {
58
76
  if (!isScheduleDefinition(schedule)) {
59
77
  throw new SyncValidationError("Sync .when(...) only accepts schedules.")
60
78
  }
61
79
  triggers.push({ type: "schedule", scheduleId: schedule.id })
62
80
  return builder
63
81
  },
64
- checkpoint<TNextCheckpoint>(): SyncBuilder<TId, TNextCheckpoint> {
82
+ checkpoint<TNextCheckpoint>(): SyncBuilder<TId, TNextCheckpoint, SyncMode> {
65
83
  return createBuilder<TNextCheckpoint>()
66
84
  },
67
85
  from<TConnector extends ConnectorDefinition>(
68
86
  connector: TConnector
69
- ): SyncReadBuilder<TId, TConnector, TCheckpoint> {
87
+ ): SyncReadBuilder<TId, TConnector, TCheckpoint, SyncMode> {
70
88
  return {
71
- read(handler): SyncTargetBuilder<TId, TConnector, TCheckpoint> {
89
+ read(handler): SyncTargetBuilder<TId, TConnector, TCheckpoint, SyncMode> {
72
90
  return {
73
91
  intoDataset(
74
92
  dataset: DatasetDefinition
75
- ): SyncDefinition<TId, TConnector, TCheckpoint> {
76
- assertDataset(dataset)
93
+ ): SyncDefinition<TId, TConnector, TCheckpoint, SyncMode> {
94
+ assertDataset(dataset, config.mode)
77
95
 
78
96
  return {
79
97
  kind: "sync",
@@ -13,6 +13,7 @@ export type {
13
13
  SyncBlobContext,
14
14
  SyncBuilder,
15
15
  SyncDefinition,
16
+ SyncMode,
16
17
  SyncReadBuilder,
17
18
  SyncReadContext,
18
19
  SyncReadHandler,
@@ -5,7 +5,7 @@ import {
5
5
  type ConnectorDefinition,
6
6
  isConnectorDefinition,
7
7
  } from "../connectors"
8
- import type { DatasetDefinition } from "../datasets"
8
+ import type { DatasetDefinition, DatasetPrimaryKey, MergeChange } from "../datasets"
9
9
  import { isDatasetDefinition } from "../datasets"
10
10
  import type { Logger } from "../logging"
11
11
  import type { ScheduleDefinition, ScheduleReference } from "../schedules"
@@ -28,18 +28,27 @@ export type SyncReadContext<TCheckpoint = never> = {
28
28
  setCheckpoint(next: TCheckpoint): void
29
29
  })
30
30
 
31
+ export type SyncMode = "snapshot" | "append" | "merge"
32
+
33
+ type MergeSyncChange = MergeChange<
34
+ Readonly<Record<string, unknown>>,
35
+ Readonly<Record<string, unknown>>
36
+ >
37
+
31
38
  /** A sync read handler may return one item, a sync iterable, or an async iterable. */
32
- export type SyncReadResult = unknown | Iterable<unknown> | AsyncIterable<unknown>
39
+ export type SyncReadResult<TMode extends SyncMode = SyncMode> = TMode extends "merge"
40
+ ? MergeSyncChange | Iterable<MergeSyncChange> | AsyncIterable<MergeSyncChange>
41
+ : unknown | Iterable<unknown> | AsyncIterable<unknown>
33
42
 
34
43
  /** User-facing batch sync options accepted by `defineSync(...)`. */
35
- export interface BatchSyncConfig {
36
- readonly mode?: "snapshot" | "append"
44
+ export interface BatchSyncConfig<TMode extends SyncMode = SyncMode> {
45
+ readonly mode?: TMode
37
46
  }
38
47
 
39
48
  /** Normalized batch sync config stored on the final sync definition. */
40
- export interface BatchSyncDefinitionConfig {
49
+ export interface BatchSyncDefinitionConfig<TMode extends SyncMode = SyncMode> {
41
50
  readonly kind: "batch"
42
- readonly mode: "snapshot" | "append"
51
+ readonly mode: TMode
43
52
  }
44
53
 
45
54
  /** V1 syncs always target a named raw dataset. */
@@ -56,11 +65,15 @@ export interface DatasetSyncTarget {
56
65
  * builder still gives users precise `context.checkpoint` and `setCheckpoint(...)` types when
57
66
  * they author a sync.
58
67
  */
59
- export type SyncReadHandler<TAdapter extends ConnectorAdapter, TCheckpoint = never> = {
68
+ export type SyncReadHandler<
69
+ TAdapter extends ConnectorAdapter,
70
+ TCheckpoint = never,
71
+ TMode extends SyncMode = SyncMode,
72
+ > = {
60
73
  bivarianceHack(
61
74
  client: ConnectorClient<TAdapter>,
62
75
  context: SyncReadContext<TCheckpoint>
63
- ): SyncReadResult | Promise<SyncReadResult>
76
+ ): SyncReadResult<TMode> | Promise<SyncReadResult<TMode>>
64
77
  }["bivarianceHack"]
65
78
 
66
79
  /**
@@ -73,13 +86,14 @@ export interface SyncDefinition<
73
86
  TId extends string = string,
74
87
  TConnector extends ConnectorDefinition = ConnectorDefinition,
75
88
  TCheckpoint = unknown,
89
+ TMode extends SyncMode = SyncMode,
76
90
  > {
77
91
  readonly kind: "sync"
78
92
  readonly id: TId
79
- readonly config: BatchSyncDefinitionConfig
93
+ readonly config: BatchSyncDefinitionConfig<TMode>
80
94
  readonly triggers: readonly ScheduleReference[]
81
95
  readonly connector: TConnector
82
- readonly read: SyncReadHandler<TConnector["adapter"], TCheckpoint>
96
+ readonly read: SyncReadHandler<TConnector["adapter"], TCheckpoint, TMode>
83
97
  readonly target: DatasetSyncTarget
84
98
  }
85
99
 
@@ -87,26 +101,36 @@ export interface SyncTargetBuilder<
87
101
  TId extends string = string,
88
102
  TConnector extends ConnectorDefinition = ConnectorDefinition,
89
103
  TCheckpoint = never,
104
+ TMode extends SyncMode = SyncMode,
90
105
  > {
91
- intoDataset(dataset: DatasetDefinition): SyncDefinition<TId, TConnector, TCheckpoint>
106
+ intoDataset<TDataset extends DatasetDefinition>(
107
+ dataset: TMode extends "merge"
108
+ ? TDataset & { readonly primaryKey: DatasetPrimaryKey }
109
+ : TDataset
110
+ ): SyncDefinition<TId, TConnector, TCheckpoint, TMode>
92
111
  }
93
112
 
94
113
  export interface SyncReadBuilder<
95
114
  TId extends string = string,
96
115
  TConnector extends ConnectorDefinition = ConnectorDefinition,
97
116
  TCheckpoint = never,
117
+ TMode extends SyncMode = SyncMode,
98
118
  > {
99
119
  read(
100
- handler: SyncReadHandler<TConnector["adapter"], TCheckpoint>
101
- ): SyncTargetBuilder<TId, TConnector, TCheckpoint>
120
+ handler: SyncReadHandler<TConnector["adapter"], TCheckpoint, TMode>
121
+ ): SyncTargetBuilder<TId, TConnector, TCheckpoint, TMode>
102
122
  }
103
123
 
104
- export interface SyncBuilder<TId extends string = string, TCheckpoint = never> {
105
- when(schedule: ScheduleDefinition): SyncBuilder<TId, TCheckpoint>
106
- checkpoint<TNextCheckpoint>(): SyncBuilder<TId, TNextCheckpoint>
124
+ export interface SyncBuilder<
125
+ TId extends string = string,
126
+ TCheckpoint = never,
127
+ TMode extends SyncMode = SyncMode,
128
+ > {
129
+ when(schedule: ScheduleDefinition): SyncBuilder<TId, TCheckpoint, TMode>
130
+ checkpoint<TNextCheckpoint>(): SyncBuilder<TId, TNextCheckpoint, TMode>
107
131
  from<TConnector extends ConnectorDefinition>(
108
132
  connector: TConnector
109
- ): SyncReadBuilder<TId, TConnector, TCheckpoint>
133
+ ): SyncReadBuilder<TId, TConnector, TCheckpoint, TMode>
110
134
  }
111
135
 
112
136
  /** Runtime type guard for values discovered from `syncs/` modules. */
@@ -120,7 +144,9 @@ export function isSyncDefinition(value: unknown): value is SyncDefinition {
120
144
  typeof value.id === "string" &&
121
145
  isRecord(value.config) &&
122
146
  value.config.kind === "batch" &&
123
- (value.config.mode === "snapshot" || value.config.mode === "append") &&
147
+ (value.config.mode === "snapshot" ||
148
+ value.config.mode === "append" ||
149
+ value.config.mode === "merge") &&
124
150
  Array.isArray(value.triggers) &&
125
151
  (value.triggers as unknown[]).every((reference) => isScheduleReference(reference)) &&
126
152
  isConnectorDefinition(value.connector) &&