@mulmoclaude/core 4.1.0 → 4.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,7 @@ const require_promptSafety = require("./promptSafety-CuczAN9d.cjs");
7
7
  const require_discovery = require("./discovery-B9EESjHP.cjs");
8
8
  const require_feeds_paths = require("./feeds/paths.cjs");
9
9
  const require_skill_bridge_index = require("./skill-bridge/index.cjs");
10
+ let node_fs = require("node:fs");
10
11
  let node_path = require("node:path");
11
12
  node_path = require_rolldown_runtime.__toESM(node_path, 1);
12
13
  let node_crypto = require("node:crypto");
@@ -1558,6 +1559,26 @@ var SCHEMA_FILE = "schema.json";
1558
1559
  var SCHEMA_DOCS_FILE = "collection-skills.md";
1559
1560
  /** Cap the rejected-schema issue list so a deeply-broken schema can't flood the result. */
1560
1561
  var MAX_SCHEMA_ISSUES = 20;
1562
+ /** Cap the rows one putItems call may write. `putOneItem` validates and
1563
+ * writes one record at a time, so a large `itemsFile` holds the tool call
1564
+ * open for minutes. Over the cap the call is refused WHOLE — a truncating
1565
+ * write that reported success would leave a half-filled collection nobody
1566
+ * knows is half-filled. */
1567
+ var MAX_PUT_ITEMS = 1e3;
1568
+ /** Refuse an `itemsFile` larger than this, from `stat` and before any read.
1569
+ * The row cap alone cannot bound the work: the file has to be read and parsed
1570
+ * WHOLE before there are rows to count, so a huge blob is paid for in full
1571
+ * first. 8 MiB is far past what 1000 records need and far short of trouble. */
1572
+ var MAX_ITEMS_FILE_BYTES = 8388608;
1573
+ /** `itemsFile` is opened read-only, without following a symlink, and without
1574
+ * blocking on a fifo — see `openContainedItemsFile`.
1575
+ *
1576
+ * `O_NOFOLLOW` and `O_NONBLOCK` are POSIX-only: on Windows they are absent, and
1577
+ * `x | undefined` is `x`, so the flags silently soften to a plain read-only
1578
+ * open. They are hardening where they exist, never the guarantee — the symlink
1579
+ * refusal is an explicit `lstat` (`verifyOpenedItemsFile`) so it holds on every
1580
+ * platform. `?? 0` states that rather than leaving it to coercion. */
1581
+ var OPEN_ITEMS_FILE_FLAGS = node_fs.constants.O_RDONLY | (node_fs.constants.O_NOFOLLOW ?? 0) | (node_fs.constants.O_NONBLOCK ?? 0);
1561
1582
  /** The workspace help-docs dir both hosts seed (`@mulmoclaude/core/workspace-setup`
1562
1583
  * syncs the bundled assets here) — the user-editable copy schemaDocs prefers. */
1563
1584
  var HELPS_DIR = "config/helps";
@@ -1758,13 +1779,167 @@ async function handleQueryItems(collection, queryArg, deps) {
1758
1779
  rows
1759
1780
  });
1760
1781
  }
1782
+ /** Rewrite a sandbox-mount prefix to the host's workspace root, so the path the
1783
+ * agent wrote to and the file this process reads are the same bytes. Anything
1784
+ * not under the mount is returned untouched — a host with no sandbox hands the
1785
+ * agent real paths already. */
1786
+ function toHostWorkspacePath(absPath, sandboxRoot, workspaceRoot) {
1787
+ if (!sandboxRoot) return absPath;
1788
+ if (absPath === sandboxRoot) return workspaceRoot;
1789
+ const prefix = sandboxRoot.endsWith("/") ? sandboxRoot : `${sandboxRoot}/`;
1790
+ if (!absPath.startsWith(prefix)) return absPath;
1791
+ return node_path.default.join(workspaceRoot, ...absPath.slice(prefix.length).split("/"));
1792
+ }
1793
+ /** The host path an `itemsFile` names — translated out of the sandbox, and
1794
+ * required to land INSIDE the workspace.
1795
+ *
1796
+ * Containment is not tidiness. `manageCollection` is always available to a
1797
+ * sandboxed agent, and an unconstrained absolute path would turn this
1798
+ * host-side handler into a read primitive for the whole host filesystem —
1799
+ * point it at any JSON array the server user can open, store the rows, read
1800
+ * them back with `getItems`. The sandbox mounts a few app directories besides
1801
+ * the workspace, but nothing that gives the agent that reach, and the
1802
+ * workspace is where its own generated files land — so confining reads to it
1803
+ * denies the primitive without costing the feature anything.
1804
+ *
1805
+ * This is the CHEAP check, for the ordinary case and a precise message. The
1806
+ * binding one is on the opened descriptor (`verifyOpenedItemsFile`) — a path
1807
+ * checked here and read again later is a path that can change in between. */
1808
+ function resolveItemsFilePath(itemsFile, deps) {
1809
+ const root = resolveBase(deps);
1810
+ const hostPath = toHostWorkspacePath(itemsFile, deps.sandboxWorkspacePath, root);
1811
+ if (!require_discovery.isContainedInRoot(hostPath, root)) return outsideWorkspaceRefusal(require_promptSafety.defangForPrompt(itemsFile));
1812
+ return { hostPath };
1813
+ }
1814
+ function outsideWorkspaceRefusal(shown) {
1815
+ return `manageCollection: \`itemsFile\` must be inside the workspace — '${shown}' is not, and the host reads this file on your behalf. Write the generated rows under the workspace and pass that path.`;
1816
+ }
1817
+ function symlinkRefusal(shown) {
1818
+ return `manageCollection: \`itemsFile\` '${shown}' is a symbolic link. Pass the real path of a regular file inside the workspace.`;
1819
+ }
1820
+ function openItemsFileRefusal(err, shown) {
1821
+ if (require_dist.isErrorWithCode(err) && (err.code === "ELOOP" || err.code === "EMLINK")) return symlinkRefusal(shown);
1822
+ return `manageCollection: could not read \`itemsFile\` '${shown}' — ${require_promptSafety.defangForPrompt(require_dist.errorMessage(err))}. It must exist inside the workspace and be readable by the host.`;
1823
+ }
1824
+ /** Everything decided about the file, decided about the OPEN DESCRIPTOR rather
1825
+ * than about the path a second time.
1826
+ *
1827
+ * Re-`stat`ing and re-`readFile`ing the pathname would leave a TOCTOU window
1828
+ * the containment check cannot close: the caller is a sandboxed agent with
1829
+ * write access to the workspace, so it can point `rows.json` at an in-workspace
1830
+ * file, call the tool, and swap the symlink to a host file outside the mount
1831
+ * while the first `await` is pending — restoring exactly the read primitive
1832
+ * containment exists to deny. Bound to one descriptor, a swap after the open
1833
+ * changes nothing about the bytes this call goes on to read.
1834
+ *
1835
+ * The `dev`/`ino` comparison is what ties the two together: it proves the
1836
+ * descriptor's inode is the one reachable at a contained path. A hardlink
1837
+ * would satisfy it, but the agent cannot create one across the mount boundary
1838
+ * — only the workspace is mounted, and a hardlink cannot cross filesystems. */
1839
+ async function verifyOpenedItemsFile(handle, hostPath, root, shown) {
1840
+ const opened = await handle.stat();
1841
+ if (!opened.isFile()) return `manageCollection: \`itemsFile\` '${shown}' is not a regular file. It must be a JSON file holding an array of record objects.`;
1842
+ if (opened.size > 8388608) return `manageCollection: \`itemsFile\` '${shown}' is ${opened.size} bytes, over the limit of ${MAX_ITEMS_FILE_BYTES}. Nothing was read; split the rows across several files and call once per file.`;
1843
+ let real;
1844
+ let atPath;
1845
+ let link;
1846
+ try {
1847
+ link = await (0, node_fs_promises.lstat)(hostPath);
1848
+ real = await (0, node_fs_promises.realpath)(hostPath);
1849
+ atPath = await (0, node_fs_promises.stat)(real);
1850
+ } catch (err) {
1851
+ return openItemsFileRefusal(err, shown);
1852
+ }
1853
+ if (link.isSymbolicLink()) return symlinkRefusal(shown);
1854
+ if (!require_discovery.isContainedInRoot(real, root)) return outsideWorkspaceRefusal(shown);
1855
+ if (atPath.ino !== opened.ino || atPath.dev !== opened.dev) return `manageCollection: \`itemsFile\` '${shown}' changed while it was being opened. Nothing was read; write the file, then call putItems.`;
1856
+ return { size: opened.size };
1857
+ }
1858
+ /** Open the file ONCE, then prove that descriptor is the contained file.
1859
+ * `O_NOFOLLOW` refuses a symlink outright rather than resolving it, and
1860
+ * `O_NONBLOCK` keeps a fifo from parking this call on `open` itself — the
1861
+ * descriptor is what `verifyOpenedItemsFile` then judges. */
1862
+ async function openContainedItemsFile(hostPath, root, shown) {
1863
+ let handle;
1864
+ try {
1865
+ handle = await (0, node_fs_promises.open)(hostPath, OPEN_ITEMS_FILE_FLAGS);
1866
+ } catch (err) {
1867
+ return openItemsFileRefusal(err, shown);
1868
+ }
1869
+ const verified = await verifyOpenedItemsFile(handle, hostPath, root, shown);
1870
+ if (typeof verified !== "string") return {
1871
+ handle,
1872
+ size: verified.size
1873
+ };
1874
+ await handle.close();
1875
+ return verified;
1876
+ }
1877
+ /** Read the descriptor into a buffer bounded by the size that was CHECKED,
1878
+ * rather than to EOF.
1879
+ *
1880
+ * `FileHandle.readFile()` reads until end-of-file, which the size check cannot
1881
+ * bound: the agent can hold the same file open and append to it after the
1882
+ * `stat` and before the read, and because appending does not change the inode,
1883
+ * the identity check cannot see it either. A 2-byte file that passed the gate
1884
+ * can hand back gigabytes. Allocating `size + 1` makes the cap hold on the
1885
+ * bytes actually taken, whatever the file does meanwhile — and that one extra
1886
+ * byte is what detects the growth, so a file being written under us is refused
1887
+ * rather than parsed as the truncated half it would otherwise look like. */
1888
+ async function readItemsBytes(handle, size, shown) {
1889
+ const buffer = Buffer.allocUnsafe(size + 1);
1890
+ const { bytesRead } = await handle.read(buffer, 0, size + 1, 0);
1891
+ if (bytesRead > size) return `manageCollection: \`itemsFile\` '${shown}' grew while it was being read. Nothing was written; finish writing the file, then call putItems.`;
1892
+ return { raw: buffer.subarray(0, bytesRead).toString("utf-8") };
1893
+ }
1894
+ function parseItemsJson(raw, shown) {
1895
+ let parsed;
1896
+ try {
1897
+ parsed = JSON.parse(raw);
1898
+ } catch (err) {
1899
+ return `manageCollection: \`itemsFile\` '${shown}' could not be read as JSON — ${require_promptSafety.defangForPrompt(require_dist.errorMessage(err))}. It must hold a JSON array of record objects.`;
1900
+ }
1901
+ if (!isRecordArray(parsed) || parsed.length === 0) return `manageCollection: \`itemsFile\` '${shown}' must hold a non-empty JSON array of record objects.`;
1902
+ return parsed;
1903
+ }
1904
+ /** Read the rows an `itemsFile` holds. Every failure comes back as tool text,
1905
+ * never a throw: a bad path or a malformed file is something the agent can fix
1906
+ * and retry. The path it passed is what the messages name, even when the bytes
1907
+ * were read from the translated host path — it is the one the agent recognises. */
1908
+ async function readItemsFile(itemsFile, deps) {
1909
+ const shown = require_promptSafety.defangForPrompt(itemsFile);
1910
+ const resolved = resolveItemsFilePath(itemsFile, deps);
1911
+ if (typeof resolved === "string") return resolved;
1912
+ const opened = await openContainedItemsFile(resolved.hostPath, resolveBase(deps), shown);
1913
+ if (typeof opened === "string") return opened;
1914
+ let read;
1915
+ try {
1916
+ read = await readItemsBytes(opened.handle, opened.size, shown);
1917
+ } catch (err) {
1918
+ return openItemsFileRefusal(err, shown);
1919
+ } finally {
1920
+ await opened.handle.close();
1921
+ }
1922
+ return typeof read === "string" ? read : parseItemsJson(read.raw, shown);
1923
+ }
1924
+ /** The rows this call will write, from whichever source it named. The cap is
1925
+ * checked here — on the resolved rows, so it holds for `items` and `itemsFile`
1926
+ * alike — and BEFORE the first write, so an over-cap call leaves the
1927
+ * collection exactly as it found it instead of half-filled. */
1928
+ async function resolvePutRows(args, deps) {
1929
+ const rows = args.itemsFile === void 0 ? args.items : await readItemsFile(args.itemsFile, deps);
1930
+ if (typeof rows === "string") return rows;
1931
+ if (rows.length > 1e3) return `manageCollection: refused — ${rows.length} rows is over the putItems limit of ${MAX_PUT_ITEMS}. Nothing was written; split them across several calls.`;
1932
+ return rows;
1933
+ }
1761
1934
  async function handlePutItems(collection, args, deps) {
1762
1935
  const store = require_discovery.storeFor(collection, { workspaceRoot: deps.workspaceRoot });
1763
1936
  const { write } = store;
1764
1937
  if (!write) return `manageCollection: ${require_discovery.readOnlyRefusal(collection.slug)} (its records are the rows of '${collection.schema.dataSource?.path}'; edit that file to change the data).`;
1938
+ const rows = await resolvePutRows(args, deps);
1939
+ if (typeof rows === "string") return rows;
1765
1940
  const written = [];
1766
1941
  const rejected = [];
1767
- for (const record of args.items) {
1942
+ for (const record of rows) {
1768
1943
  const outcome = await putOneItem(collection, store, write, record, args.mode, deps);
1769
1944
  if (outcome.written) written.push(outcome.written);
1770
1945
  if (outcome.rejected) rejected.push(outcome.rejected);
@@ -1827,14 +2002,36 @@ function parsePutMode(mode) {
1827
2002
  function isRecordArray(value) {
1828
2003
  return require_dist.isUnknownArray(value) && value.every(require_dist.isRecord);
1829
2004
  }
2005
+ /** `items` and `itemsFile` are ALTERNATIVES, never a pair: two row sets in one
2006
+ * call has no correct reading — honouring one silently discards the other, and
2007
+ * concatenating them writes rows the caller never asked to write together. So
2008
+ * both present is refused, rather than resolved by precedence.
2009
+ *
2010
+ * A relative `itemsFile` is refused too, for a reason the agent cannot see from
2011
+ * where it stands: this tool runs inside the HOST'S SERVER PROCESS, whose
2012
+ * working directory is not the agent's. A relative path would not reliably
2013
+ * fail — it would resolve against an unrelated directory and either miss or,
2014
+ * worse, read a different file that happens to share the name. */
2015
+ function parseItemsSource(items, itemsFile) {
2016
+ if (items !== void 0 && itemsFile !== void 0) return "manageCollection: pass either `items` or `itemsFile` for putItems, not both — two row sets in one call is ambiguous.";
2017
+ if (itemsFile !== void 0) return parseItemsFile(itemsFile);
2018
+ if (!isRecordArray(items) || items.length === 0) return "manageCollection: putItems needs `items` (a non-empty array of record objects) or `itemsFile` (an absolute path to a JSON file of them — use it for a set a script generated, so the rows never pass through your context).";
2019
+ return { items };
2020
+ }
2021
+ function parseItemsFile(itemsFile) {
2022
+ const file = typeof itemsFile === "string" ? itemsFile.trim() : "";
2023
+ if (!file) return "manageCollection: `itemsFile` must be a non-empty absolute path to a JSON file of record objects.";
2024
+ if (!node_path.default.isAbsolute(file)) return `manageCollection: \`itemsFile\` must be an ABSOLUTE path — '${require_promptSafety.defangForPrompt(file)}' is relative, and this tool runs in the host's server process, whose working directory is not yours. Pass the full path.`;
2025
+ return { itemsFile: file };
2026
+ }
1830
2027
  function parsePutItems(args, slug) {
1831
- const { items, mode } = args;
1832
- if (!isRecordArray(items) || items.length === 0) return "manageCollection: `items` is required for putItems — a non-empty array of record objects.";
1833
- const putMode = parsePutMode(mode);
2028
+ const source = parseItemsSource(args.items, args.itemsFile);
2029
+ if (typeof source === "string") return source;
2030
+ const putMode = parsePutMode(args.mode);
1834
2031
  if (putMode === null) return "manageCollection: `mode` must be \"upsert\" (default), \"create\", or \"merge\".";
1835
2032
  return {
2033
+ ...source,
1836
2034
  slug,
1837
- items,
1838
2035
  mode: putMode
1839
2036
  };
1840
2037
  }
@@ -1960,7 +2157,7 @@ async function handlePutSchema(slug, schemaArg, deps) {
1960
2157
  written: true
1961
2158
  });
1962
2159
  }
1963
- var MANAGE_COLLECTION_PROMPT = "Use `manageCollection` instead of raw Read/Write/Edit when working with a collection's records OR its schema (raw file I/O stays available as the escape hatch). Before authoring or changing a collection's `schema.json`, call `schemaDocs` to load the field/DSL reference — the default reply is the core authoring guide plus a table of contents; fetch advanced sections (actions, bells, calendar/kanban views, dataSource, storage) by passing their heading as `topic` rather than dumping `topic: \"all\"`. Then read with `getSchema` and write with `putSchema` — `putSchema` validates the whole schema before writing and returns actionable errors instead of silently failing discovery's validation. `getItems` is the only way to see computed values — `derived` fields (e.g. a portfolio's value), `toggle` projections, and `embed` records are host-computed and never present in the stored JSON files. On large collections pass `ids` and/or `fields` to keep the result small. For a question that spans collections (\"which clients have unpaid invoices?\"), start with `getOntology`: it lists every collection with its primaryKey, record count, and outbound `ref`/`embed` relations, so you know which collections to join before reading any records. `putItems` validates every row against the schema before writing (required fields, enum values, primaryKey = record id) and returns `{ written, rejected }`; fix each rejected row using its `problem` text and retry just those rows. Never include computed fields in a row you write. To update a few fields of an existing record, use `mode: \"merge\"` with a partial row ({ id, <changed fields> }) — the default upsert replaces the WHOLE record, so a partial upsert would silently erase every optional field it omits. `deleteItems` removes records by id and returns `{ deleted, rejected }`; an id that doesn't exist comes back rejected rather than counted as deleted, so check `rejected` before reporting a deletion as done. Answer aggregation questions (counts, sums, averages, group-bys) with `queryItems` on ANY collection — on a dataSource (CSV) collection it scans the whole file (getItems is row-capped, so aggregates computed from its output can be silently wrong on large files); on a file-backed collection it aggregates the enriched records, so computed fields (derived/rollup/toggle) are queryable columns.";
2160
+ var MANAGE_COLLECTION_PROMPT = "Use `manageCollection` instead of raw Read/Write/Edit when working with a collection's records OR its schema (raw file I/O stays available as the escape hatch). Before authoring or changing a collection's `schema.json`, call `schemaDocs` to load the field/DSL reference — the default reply is the core authoring guide plus a table of contents; fetch advanced sections (actions, bells, calendar/kanban views, dataSource, storage) by passing their heading as `topic` rather than dumping `topic: \"all\"`. Then read with `getSchema` and write with `putSchema` — `putSchema` validates the whole schema before writing and returns actionable errors instead of silently failing discovery's validation. `getItems` is the only way to see computed values — `derived` fields (e.g. a portfolio's value), `toggle` projections, and `embed` records are host-computed and never present in the stored JSON files. On large collections pass `ids` and/or `fields` to keep the result small. For a question that spans collections (\"which clients have unpaid invoices?\"), start with `getOntology`: it lists every collection with its primaryKey, record count, and outbound `ref`/`embed` relations, so you know which collections to join before reading any records. `putItems` validates every row against the schema before writing (required fields, enum values, primaryKey = record id) and returns `{ written, rejected }`; fix each rejected row using its `problem` text and retry just those rows. Never include computed fields in a row you write. When the rows come from a script rather than from you (a generated schedule, an imported set, anything past a few dozen records), write them to a JSON file UNDER THE WORKSPACE and pass its absolute path as `itemsFile` instead of `items` — the host reads the file, so the rows never pass through your context. Do NOT hand-transcribe a generated file into `items`, and never drive the collection by spawning the MCP bridge yourself. To update a few fields of an existing record, use `mode: \"merge\"` with a partial row ({ id, <changed fields> }) — the default upsert replaces the WHOLE record, so a partial upsert would silently erase every optional field it omits. `deleteItems` removes records by id and returns `{ deleted, rejected }`; an id that doesn't exist comes back rejected rather than counted as deleted, so check `rejected` before reporting a deletion as done. Answer aggregation questions (counts, sums, averages, group-bys) with `queryItems` on ANY collection — on a dataSource (CSV) collection it scans the whole file (getItems is row-capped, so aggregates computed from its output can be silently wrong on large files); on a file-backed collection it aggregates the enriched records, so computed fields (derived/rollup/toggle) are queryable columns.";
1964
2161
  /** Validate getItems' optional `ids`/`fields` args, then delegate. */
1965
2162
  async function dispatchGetItems(collection, args, deps) {
1966
2163
  const ids = optionalStringArray(args.ids, "ids");
@@ -2051,7 +2248,11 @@ var MANAGE_COLLECTION_DEFINITION = {
2051
2248
  items: {
2052
2249
  type: "array",
2053
2250
  items: { type: "object" },
2054
- description: "putItems: the record objects to store. Each must carry the schema's primaryKey value (it doubles as the filename)."
2251
+ description: "putItems: the record objects to store, inline. Each must carry the schema's primaryKey value (it doubles as the filename). For rows a script generated, pass `itemsFile` instead — never both."
2252
+ },
2253
+ itemsFile: {
2254
+ type: "string",
2255
+ description: "putItems: an ABSOLUTE path to a JSON file holding the array of record objects, read by the host — the alternative to `items` for rows a script produced. Use it whenever the data already exists as a file: passing a generated set of several hundred records through `items` means writing every byte of it yourself. The path must be absolute (this tool runs in the host's server process, whose working directory is not yours) and must be INSIDE the workspace — write the generated file under the workspace, not to a system temp dir. The file must hold a non-empty JSON array, and `items` and `itemsFile` are mutually exclusive."
2055
2256
  },
2056
2257
  mode: {
2057
2258
  type: "string",
@@ -2087,6 +2288,18 @@ function makeManageCollectionTool(deps = {}) {
2087
2288
  };
2088
2289
  }
2089
2290
  //#endregion
2291
+ Object.defineProperty(exports, "MAX_ITEMS_FILE_BYTES", {
2292
+ enumerable: true,
2293
+ get: function() {
2294
+ return MAX_ITEMS_FILE_BYTES;
2295
+ }
2296
+ });
2297
+ Object.defineProperty(exports, "MAX_PUT_ITEMS", {
2298
+ enumerable: true,
2299
+ get: function() {
2300
+ return MAX_PUT_ITEMS;
2301
+ }
2302
+ });
2090
2303
  Object.defineProperty(exports, "MAX_RECORD_ISSUES", {
2091
2304
  enumerable: true,
2092
2305
  get: function() {
@@ -2304,4 +2517,4 @@ Object.defineProperty(exports, "validateRecordObject", {
2304
2517
  }
2305
2518
  });
2306
2519
 
2307
- //# sourceMappingURL=server-DSrRdPvf.cjs.map
2520
+ //# sourceMappingURL=server-meS6kjBN.cjs.map