@lotics/cli 0.97.0 → 0.99.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/src/cli.js CHANGED
@@ -44165,9 +44165,7 @@ import os2 from "node:os";
44165
44165
  import path7 from "node:path";
44166
44166
  import readline from "node:readline";
44167
44167
 
44168
- // src/client.ts
44169
- import fs from "node:fs";
44170
- import path from "node:path";
44168
+ // ../shared/src/transport_error.ts
44171
44169
  function gatewayErrorMessage(status) {
44172
44170
  if (status === 524) {
44173
44171
  return "The request took too long to finish (gateway timeout). It may still be running \u2014 check back in a moment, or try again.";
@@ -44181,6 +44179,10 @@ function transportErrorMessage(status, parsed) {
44181
44179
  const jsonMessage = parsed && typeof parsed.message === "string" ? parsed.message : null;
44182
44180
  return parsed === null || status >= 500 || jsonMessage === null ? gatewayErrorMessage(status) : jsonMessage;
44183
44181
  }
44182
+
44183
+ // src/client.ts
44184
+ import fs from "node:fs";
44185
+ import path from "node:path";
44184
44186
  function findAvailableFilename(dir, filename, reserved) {
44185
44187
  const isTaken = (name) => {
44186
44188
  const full = path.join(dir, name);
@@ -64415,7 +64417,8 @@ var aggregateOperationSchema = external_exports.enum([
64415
64417
  "checked",
64416
64418
  "unchecked",
64417
64419
  "percent_checked",
64418
- "percent_unchecked"
64420
+ "percent_unchecked",
64421
+ "string_agg"
64419
64422
  ]);
64420
64423
  var OPERATION_FIELD_TYPES = {
64421
64424
  empty: ["text", "number", "date", "select", "select_member", "select_record_link", "files", "formula", "rollup"],
@@ -64483,6 +64486,9 @@ var queryOutputColumnSchema = zod_default.object({
64483
64486
  ),
64484
64487
  source_field_key: zod_default.string().optional().describe(
64485
64488
  "Source field this column originates from. Populated alongside source_table_id for passthrough columns."
64489
+ ),
64490
+ computed: zod_default.boolean().optional().describe(
64491
+ "True when the value is DERIVED rather than a passthrough of the addressed field \u2014 a link extraction reads a field on the link TARGET, so the row's own source record is a different table. Such a column may still carry source addressing (it is what resolves labels and option sets), but that addressing is NOT a write target: writable_target is refused on a computed column and on anything that passes one through."
64486
64492
  )
64487
64493
  });
64488
64494
  var queryOutputSchemaSchema = zod_default.object({
@@ -64530,12 +64536,21 @@ var queryProjectionColumnSchema = zod_default.union([
64530
64536
  zod_default.string().describe("Passthrough shorthand: a source field key. Output = the key, type = the field's type."),
64531
64537
  queryProjectionColumnObjectSchema
64532
64538
  ]);
64539
+ var STRING_AGG_MAX_VALUES = 100;
64540
+ var STRING_AGG_DEFAULT_MAX_VALUES = 20;
64533
64541
  var queryAggregateColumnSchema = zod_default.object({
64534
64542
  output: zod_default.string().describe("Output column name"),
64535
64543
  type: queryColumnTypeSchema.describe("Output column type"),
64536
64544
  operation: aggregateOperationSchema,
64537
64545
  input_column: zod_default.string().optional().describe(
64538
64546
  "Input column to aggregate. Required for non-count operations; ignored for count."
64547
+ ),
64548
+ separator: zod_default.string().max(8).optional().describe('string_agg only. Joins the values. Defaults to ", ".'),
64549
+ distinct: zod_default.boolean().optional().describe(
64550
+ "string_agg only. Collapses repeats \u2014 a group of containers sized 40HC/40HC/20DC aggregates to '20DC, 40HC'. Defaults to true, which is almost always what a summary column wants."
64551
+ ),
64552
+ max_values: zod_default.number().int().min(1).max(STRING_AGG_MAX_VALUES).optional().describe(
64553
+ `string_agg only. Caps how many values are emitted so an unbounded child set cannot produce a giant cell. Defaults to ${STRING_AGG_DEFAULT_MAX_VALUES}. Pair with a \`unique\` aggregate to render an accurate '+N more'.`
64539
64554
  )
64540
64555
  });
64541
64556
  var rankingWindowFunctionSchema = (fn) => zod_default.object({ output: zod_default.string().describe("Output column name"), fn: zod_default.literal(fn) }).strict();
@@ -64892,6 +64907,12 @@ var appAgentDeclarationSchema = zod_default.object({
64892
64907
  knowledge_doc_ids: zod_default.array(zod_default.string().min(1)).optional().describe(
64893
64908
  "Knowledge docs the agent may read, validated at declare time against the app owner's `use` access. Small docs are inlined into the agent's system prompt each run; a doc too large to inline requires the code tools (code_exec) in tool_names and is read by staging it into a code run. Omit for an agent that needs no knowledge."
64894
64909
  ),
64910
+ query_aliases: zod_default.array(zod_default.string().min(1)).optional().describe(
64911
+ "Named queries from this app's manifest the agent may run via `run_app_query`, validated at declare time against the app's own queries. This is the agent's ENTIRE read surface over workspace data \u2014 a template fixes the tables, filters, and projection, so it bounds rows and columns, not just tables. Omit for an agent that reads no records."
64912
+ ),
64913
+ workflow_aliases: zod_default.array(zod_default.string().min(1)).optional().describe(
64914
+ "Workflows from this app's manifest the agent may invoke via `run_app_workflow`, validated at declare time against the app's own workflows. This is the agent's ENTIRE write surface \u2014 the same declared mutation path the app's UI uses, so table hooks and side-effect harvesting apply. Omit for a read-only agent."
64915
+ ),
64895
64916
  model_id: zod_default.string().min(1).optional().describe(
64896
64917
  "Chat model id the agent runs on, validated against ACCEPTED_CHAT_MODEL_IDS. Omit to follow the platform default chat model, resolved at run time \u2014 the preferred choice: the agent tracks model generations with no per-app rewrite. Pin only a deliberate, tested choice."
64897
64918
  ),
@@ -1,12 +1,4 @@
1
1
  import type { ContractConfigEntry, KnowledgeUpgradeEntry, ModifiedArtifact, PackageContentBinding, UpgradeResolutions } from "./package_content_types";
2
- /**
3
- * The error message for a non-ok response. A genuine JSON error (a 4xx carrying
4
- * a `message`) surfaces verbatim; a non-JSON body (a gateway HTML page), any
5
- * 5xx, or a JSON body without a `message` falls back to a body-free,
6
- * status-derived message. `parsed` is the JSON.parse of the body, or `null`.
7
- * In parity (by value, no shared dep) with `packages/app-sdk/src/rpc.ts`.
8
- */
9
- export declare function transportErrorMessage(status: number, parsed: unknown): string;
10
2
  /** One sort key forwarded to the app query RPC (wire shape of a `TableRecordSort` entry). */
11
3
  export interface AppQuerySortKey {
12
4
  field_key: string;
@@ -89,19 +81,20 @@ export interface ToolInfo {
89
81
  }
90
82
  /**
91
83
  * A single knowledge doc with its HYDRATED body — the shape of
92
- * `GET /v1/knowledge_docs/{id}`. `content` is resolved server-side from the
93
- * doc's content file (or the parked column for a legacy row), so this is the
94
- * one content-read path a non-sandbox client has. `content_file_id` is the
95
- * concurrency token the REST PATCH echoes; the `update_knowledge` TOOL CASes
96
- * internally, so a CLI caller never needs to pass it.
84
+ * `GET /v1/knowledge_docs/{id}`, and the one content-read path a non-sandbox
85
+ * client has. `content_sha` is the concurrency token the REST PATCH echoes back
86
+ * as `expected_content_sha`; the `update_knowledge` TOOL CASes internally, so a
87
+ * CLI caller never needs to pass it.
97
88
  */
98
89
  export interface KnowledgeDocDetail {
99
90
  id: string;
100
91
  workspace_id: string;
101
92
  name: string;
102
93
  description: string;
94
+ /** Folder address; "" is root. */
95
+ path: string;
103
96
  content: string;
104
- content_file_id: string | null;
97
+ content_sha: string | null;
105
98
  files: unknown[];
106
99
  created_at: string;
107
100
  updated_at: string;
@@ -1,36 +1,6 @@
1
+ import { transportErrorMessage } from "@lotics/shared/transport_error";
1
2
  import fs from "node:fs";
2
3
  import path from "node:path";
3
- /**
4
- * A user-facing message for a transport/gateway failure — derived from the HTTP
5
- * status, never from the response body. A 524 (edge timeout on a long run), any
6
- * 5xx, or a non-JSON body (an HTML error page) must NOT surface its raw body. In
7
- * parity (by value, no shared dep) with the published transport in
8
- * `packages/app-sdk/src/rpc.ts`.
9
- */
10
- function gatewayErrorMessage(status) {
11
- if (status === 524) {
12
- return "The request took too long to finish (gateway timeout). It may still be running — check back in a moment, or try again.";
13
- }
14
- if (status >= 500) {
15
- return "The service is temporarily unavailable. Please try again shortly.";
16
- }
17
- return "The service returned an unexpected response. Please try again.";
18
- }
19
- /**
20
- * The error message for a non-ok response. A genuine JSON error (a 4xx carrying
21
- * a `message`) surfaces verbatim; a non-JSON body (a gateway HTML page), any
22
- * 5xx, or a JSON body without a `message` falls back to a body-free,
23
- * status-derived message. `parsed` is the JSON.parse of the body, or `null`.
24
- * In parity (by value, no shared dep) with `packages/app-sdk/src/rpc.ts`.
25
- */
26
- export function transportErrorMessage(status, parsed) {
27
- const jsonMessage = parsed && typeof parsed.message === "string"
28
- ? parsed.message
29
- : null;
30
- return parsed === null || status >= 500 || jsonMessage === null
31
- ? gatewayErrorMessage(status)
32
- : jsonMessage;
33
- }
34
4
  function findAvailableFilename(dir, filename, reserved) {
35
5
  // `reserved` tracks absolute paths claimed by in-flight downloads in the same
36
6
  // batch — required for parallel callers because the file may not be on disk
@@ -7,8 +7,8 @@ them on demand** while it works, pulling in only the lines it needs.
7
7
 
8
8
  They are deliberately **not** injected into the agent wholesale. Bulk-loading every doc into
9
9
  every request would burn the context budget and drown the signal. Instead the agent retrieves
10
- from them through a search funnel, so a 10,000-line tariff book costs nothing until a question
11
- actually touches it.
10
+ from them line by line — grep, then read — so a 10,000-line tariff book costs nothing until a
11
+ question actually touches it, and a 10MB one is no different.
12
12
 
13
13
  This is a **capability + usage** guide. For the exact input schema of any tool named here, run
14
14
  `lotics tools <tool_name>`.
@@ -34,9 +34,9 @@ This reads the body from your filesystem and creates the doc through `create_kno
34
34
  `content` from a file or stdin. Either way, `create_knowledge` takes three things:
35
35
 
36
36
  - `name` — what it is.
37
- - `description` — what the doc is **and how to retrieve from it**: its vocabulary, synonyms for
38
- colloquial terms, and what to read for detail. This is surfaced when the doc is *found*,
39
- before its content is read, so the agent knows whether to open it.
37
+ - `description` — what the doc is **and what to grep it for**: its vocabulary and the synonyms
38
+ an ambiguous query would use. Surfaced in the tree before any body is read, and truncated at
39
+ 200 characters. See *Write for grep* below.
40
40
  - `content` — Markdown.
41
41
 
42
42
  A new doc is created **owned by you, active in your own agent context, and private** — no one
@@ -61,28 +61,84 @@ So: shared + active → the agent can find and read it. Shared but deactivated
61
61
  that member's agent. This is the token economy in action — activation is how a member curates
62
62
  which rulebooks their agent carries.
63
63
 
64
- ## How an agent uses a doc — catalog, then read
64
+ ## How an agent uses a doc — ls, grep, cat
65
+
66
+ Docs are not injected wholesale. The agent works the corpus like a filesystem, and all three
67
+ verbs respect access + activation, so only docs the caller may use ever surface.
68
+
69
+ 1. **`list_knowledge`** — `ls`. The corpus as a folder tree: name, id, folder, and a truncated
70
+ description. No bodies.
71
+ 2. **`grep_knowledge`** — `grep -rn`. Substring match across every readable doc (or one doc, or
72
+ one folder), returning **doc, line number, and the matching line**. It runs inside Postgres,
73
+ so only matching lines cross the wire.
74
+ 3. **`read_knowledge`** — `cat` / `sed -n 'X,Yp'`. Read a doc whole or by line range, to see
75
+ the context around a hit.
76
+
77
+ Matching is **substring, not ranked** — there is no index and no tokenizer, which is why a
78
+ corpus in any script works and why nothing goes stale. The consequence is that the *agent*
79
+ does the narrowing: a distinctive phrase is sharply selective, a whole question matches
80
+ everything. `grep_knowledge` always reports `total_matches`, so "too broad" is visible and
81
+ cheap to fix.
82
+
83
+ ### Matching options
84
+
85
+ - **Diacritics fold by default** — `ca phe` matches `cà phê`. `diacritic_insensitive: false` matches
86
+ tone marks exactly.
87
+ - **Case folds by default**, independently of diacritics. `case_sensitive: true` matches case
88
+ exactly — useful for an acronym (`NK` vs `nk`) that a folded search would blur.
89
+ - **Whitespace is normalized on both sides.** A body converted from PDF, Word or Excel carries
90
+ non-breaking spaces, soft hyphens, zero-width marks and padded runs that nobody types into a
91
+ query; those fold to ordinary single spaces before matching, so a correct search does not return
92
+ a silent zero on text that is present. Your pattern is normalized the same way.
93
+ - **`regex: true`** treats the pattern as a POSIX regular expression: quantifiers, character
94
+ classes, alternation, anchors, and **`\y` for a word boundary**. Note `\y`, not `\b` — Postgres
95
+ spells it differently, and `\b` is rewritten for you rather than silently matching nothing.
96
+ Literal is the default on purpose: `0901.11.20` as a regex would also match `0901X11Y20`.
65
97
 
66
- Docs are not injected wholesale. The agent finds the right doc from a **catalog**, then reads
67
- only what it needs. Both steps respect access + activation, so only docs the caller may use ever
68
- surface.
69
-
70
- 1. **Catalog** — `list_knowledge` returns every usable doc as `{ id, name, description }` — no
71
- bodies. This is why the *description* carries the weight: it is what the agent reads before
72
- deciding to open a doc. Write it to sell the doc — its vocabulary, the colloquial synonyms an
73
- ambiguous query would use, and what it covers.
74
- 2. **Read** — the chat agent stages the chosen doc's content file into a code run and greps it
75
- there (the body arrives as a file to `cat`/`grep`). Over this CLI you read a body directly
76
- with `lotics knowledge get <id>`.
77
-
78
- Structure your content so a reader lands on the answer without loading the rest:
98
+ ```
99
+ grep_knowledge({ pattern: "\\yNK\\y", regex: true, case_sensitive: true })
100
+ ```
79
101
 
80
- - Organize under clear Markdown headers.
81
- - Keep each searchable unit self-contained — a section for prose, **one record per line** for
82
- dense/tabular data (a price row, a code entry) — carrying both the terms someone would search
83
- for and its answer.
84
- - Lead with the most-queried fields.
85
- - Note colloquial synonyms next to official terms, so an ambiguous query still matches.
102
+ ### What a result is bounded by
103
+
104
+ A result carries at most 40.000 characters. Matched lines are admitted first and context fills
105
+ what remains, so an answer is never dropped to make room for a neighbouring line. Anything the
106
+ budget cut is REPORTED — `chars_elided_matches` (an answer was dropped) and
107
+ `context_omitted_matches` (an answer was kept without the context you asked for) mean different
108
+ things and call for different fixes: narrow the pattern, or ask for fewer `context_lines`.
109
+ Individual lines longer than 600 characters are clipped and marked, with `read_knowledge` giving
110
+ the rest.
111
+
112
+ `total_matches` is always the true total, even when fewer are shown.
113
+
114
+ ## Write for grep — the rules that decide whether an answer is findable
115
+
116
+ Retrieval addresses **lines**. Every rule below follows from that one fact, and ignoring them
117
+ is the difference between a doc that answers and a doc that merely exists.
118
+
119
+ - **One self-contained fact per line.** A grep hit returns *that line*. For dense or tabular
120
+ data — a price row, a tariff code, a charge entry — put the whole record on one line.
121
+ - **Repeat the searchable terms on every line; headings do not carry down.** A line reading
122
+ `Rate: 15%` under a heading `Roasted coffee` will never match a search for `coffee`. Restate
123
+ the identifying terms inline, even when it reads redundantly to a human. This is the single
124
+ rule most often got wrong.
125
+ - **Keep a line under ~600 characters.** Past that the line is truncated in the agent's view
126
+ and marked as cut; the agent can still `read_knowledge` for the rest, but it costs a round
127
+ trip. Put the identifying terms early and the long tail late.
128
+ - **Put synonyms and translations on the line itself.** A bilingual row (`Cà phê, đã rang /
129
+ Coffee, roasted`) matches queries in either language for free. The same trick carries
130
+ colloquial terms next to official ones.
131
+ - **For prose, keep a rule and its exception close.** A hit returns one line, so a rule on line
132
+ 40 and its exception on line 90 can be retrieved apart. Use `context_lines` when reading, and
133
+ keep related clauses adjacent when writing.
134
+
135
+ Markdown headings are ordinary text — useful for a human and for orienting a `read_knowledge`,
136
+ but they carry no retrieval weight of their own.
137
+
138
+ **The description is the discovery hint.** It is what the agent sees in the tree before it has
139
+ read a byte of the body, so it is the only clue for *what term to grep for*. Write it to name
140
+ the doc's vocabulary — the words that actually appear inside it. Keep it to a sentence: it is
141
+ truncated at 200 characters, and a keyword-stuffed description is cut, not rewarded.
86
142
 
87
143
  ## Updating a doc — `lotics knowledge update`
88
144
 
@@ -95,8 +151,8 @@ Send only the fields you're changing. `--from` / `--content` replaces the body;
95
151
  re-reads the current content pointer and version-chains the new body — so there is no version
96
152
  token to pass from the CLI. (The chat agent may instead send an `edits` array — anchored
97
153
  replace / insert / append — for a surgical change; see `lotics tools update_knowledge`.) Refine
98
- structure as you learn what users actually ask: add the synonym that failed to match, split the
99
- section that was too coarse to read.
154
+ structure as you learn what users actually ask: add the synonym that failed to match, split a
155
+ line that was too coarse, restate a term the heading was carrying.
100
156
 
101
157
  ## Package-managed knowledge
102
158
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lotics/cli",
3
- "version": "0.97.0",
3
+ "version": "0.99.1",
4
4
  "description": "Lotics SDK and CLI for AI agents",
5
5
  "type": "module",
6
6
  "bin": {