openmausbot 0.1.74 → 0.1.75

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,9 +10,18 @@
10
10
  // ask_user — the agent can pose a question mid-run and wait; the
11
11
  // human's words come back verbatim.
12
12
  //
13
+ // One tool arrives through `approve` that is not a permission at all: the
14
+ // CLI's own AskUserQuestion. It is the tool the model actually reaches for
15
+ // when it wants a person to choose, and acceptEdits will not run it unasked,
16
+ // so it lands here looking like "may I run a tool?" — which is how a question
17
+ // ended up on screen as an Allow/Deny box over a JSON blob. It is intercepted
18
+ // below and asked as what it is: one card carrying every question it posed,
19
+ // answered through the tool's own `answers` field.
20
+ //
13
21
  // stdout is the MCP channel — never console.log here.
14
22
  import { connect } from "node:net";
15
23
  import { randomUUID } from "node:crypto";
24
+ import { parseAskQuestions, questionAnswersByQuestion } from "../shared/ask-question.js";
16
25
  const socketPath = process.argv[2] ?? "";
17
26
  const waiting = new Map();
18
27
  const conn = connect(socketPath);
@@ -45,6 +54,77 @@ conn.on("data", (chunk) => {
45
54
  }
46
55
  });
47
56
  const send = (obj) => process.stdout.write(JSON.stringify(obj) + "\n");
57
+ /** Hand one ask to the broker and wait for the human's answer. */
58
+ function askBroker(ask) {
59
+ return new Promise((resolve) => {
60
+ waiting.set(String(ask.id), resolve);
61
+ if (conn.destroyed)
62
+ return dead();
63
+ try {
64
+ conn.write(JSON.stringify(ask) + "\n");
65
+ }
66
+ catch {
67
+ dead();
68
+ }
69
+ });
70
+ }
71
+ // ── the CLI's own AskUserQuestion ──────────────────────────────────────
72
+ const ASK_USER_QUESTION = "AskUserQuestion";
73
+ const asRecord = (value) => typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
74
+ /** One question, in the shape the broker and the option card understand. */
75
+ /**
76
+ * Ask the person every question at once, then answer the tool the way it
77
+ * documents.
78
+ *
79
+ * The answer is NOT a bare allow. A bare allow tells the CLI to go and run
80
+ * AskUserQuestion, and a headless run has no dialog to collect anything —
81
+ * the model gets "The user did not answer the questions." back and the
82
+ * click is thrown away. The `answers` object is the field the tool's own
83
+ * schema calls "User answers collected by the permission component":
84
+ * keyed by the question's text, one comma-joined string per question.
85
+ *
86
+ * One ask, not one per question: the card draws the whole set with a tab
87
+ * each, so a person sees what they are committing to before answering any
88
+ * of it. `questionAnswersByQuestion` maps the answer text back per question.
89
+ *
90
+ * There is deliberately no "fall through to the permission path" here. That
91
+ * path cannot answer this tool — allowing it throws the click away, as
92
+ * above — so the fallback offered a person an Allow/Deny box over raw JSON.
93
+ * That box is the bug this file exists to remove.
94
+ */
95
+ async function answerNativeQuestions(input) {
96
+ // Unanswerable entries are SKIPPED by the parser, not fatal: one bad entry
97
+ // must not cost the user the good questions beside it. Nothing left means
98
+ // the whole call was unanswerable, which becomes a denial.
99
+ const questions = parseAskQuestions(input);
100
+ if (!questions?.length) {
101
+ return JSON.stringify({
102
+ behavior: "deny",
103
+ message: "OpenMausBot: this AskUserQuestion call had no answerable question (each one needs question text), so nobody was shown it. Ask again with a well-formed call, or continue without it.",
104
+ });
105
+ }
106
+ const answer = await askBroker({
107
+ t: "ask",
108
+ id: randomUUID(),
109
+ kind: "question",
110
+ tool: ASK_USER_QUESTION,
111
+ input: { questions },
112
+ });
113
+ // A question is only ever denied when the broker is gone.
114
+ if (answer.behavior === "deny") {
115
+ return JSON.stringify({ behavior: "deny", message: answer.message || "Denied from OpenMausBot" });
116
+ }
117
+ // What lands in `answers` turns on WHO answered, not on whether there are
118
+ // words. The broker's own notes are words — the timeout's "nobody answered
119
+ // in time, use your best judgment" is a whole sentence — so filing anything
120
+ // non-blank hands the model system text in the slot reserved for what the
121
+ // person chose. Left out, the CLI reports the question as unanswered, which
122
+ // is the truth.
123
+ const answers = answer.source === "user" && typeof answer.message === "string"
124
+ ? questionAnswersByQuestion(answer.message, questions)
125
+ : {};
126
+ return JSON.stringify({ behavior: "allow", updatedInput: { ...asRecord(input), answers } });
127
+ }
48
128
  const TOOLS = [
49
129
  {
50
130
  name: "approve",
@@ -93,6 +173,12 @@ async function handle(msg) {
93
173
  if (msg.method === "tools/call") {
94
174
  const name = msg.params?.name;
95
175
  const args = msg.params?.arguments ?? {};
176
+ const reply = (text) => send({ jsonrpc: "2.0", id: msg.id, result: { content: [{ type: "text", text }] } });
177
+ // AskUserQuestion is a question wearing a permission's clothes. It never
178
+ // continues into the permission path below — see nativeQuestions.
179
+ if (name === "approve" && args.tool_name === ASK_USER_QUESTION) {
180
+ return reply(await answerNativeQuestions(args.input));
181
+ }
96
182
  const askId = randomUUID();
97
183
  const isQuestion = name === "ask_user";
98
184
  // the CLI may include its own suggested permission rules; on allow we
@@ -103,20 +189,9 @@ async function handle(msg) {
103
189
  : Array.isArray(args.suggestions)
104
190
  ? args.suggestions
105
191
  : null;
106
- const answer = await new Promise((resolve) => {
107
- waiting.set(askId, resolve);
108
- if (conn.destroyed)
109
- return dead();
110
- const ask = isQuestion
111
- ? { t: "ask", id: askId, kind: "question", tool: "ask_user", input: { question: args.question, choices: args.choices } }
112
- : { t: "ask", id: askId, tool: args.tool_name, input: args.input };
113
- try {
114
- conn.write(JSON.stringify(ask) + "\n");
115
- }
116
- catch {
117
- dead();
118
- }
119
- });
192
+ const answer = await askBroker(isQuestion
193
+ ? { t: "ask", id: askId, kind: "question", tool: "ask_user", input: { question: args.question, choices: args.choices } }
194
+ : { t: "ask", id: askId, tool: args.tool_name, input: args.input });
120
195
  let text = answer.message || "No answer was given — use your best judgment.";
121
196
  if (!isQuestion) {
122
197
  if (answer.behavior === "allow") {
@@ -129,7 +204,7 @@ async function handle(msg) {
129
204
  text = JSON.stringify({ behavior: "deny", message: answer.message || "Denied from OpenMausBot" });
130
205
  }
131
206
  }
132
- return send({ jsonrpc: "2.0", id: msg.id, result: { content: [{ type: "text", text }] } });
207
+ return reply(text);
133
208
  }
134
209
  if (String(msg.method ?? "").startsWith("notifications/"))
135
210
  return;
@@ -1,5 +1,5 @@
1
- // Setup mode: the coaching block a bot gets when it has not been set up yet,
2
- // or when the user asks for it with /setup. The bot interviews the user, says
1
+ // Setup mode: optional coaching when the user asks for it with /setup.
2
+ // An empty profile is not a request to interview the user. The bot says
3
3
  // what it intends, and then configures itself only through proposal cards
4
4
  // (propose_profile, propose_routine, skill_manage, request_credential) — so
5
5
  // nothing changes without the user's approval. Mirrors skill-learn.ts:
@@ -29,11 +29,10 @@ export function expandSetupTurnText(userText) {
29
29
  ? `Set yourself up for this job: ${setup.request}`
30
30
  : "Set yourself up. Ask me what you need to know, then propose your configuration.";
31
31
  }
32
- /** A bot with neither standing instructions nor a description has not been
33
- * set up. /setup re-enters the mode for a configured bot. */
32
+ /** Only an explicit setup request enters coaching. Existing/blank bots must
33
+ * still do ordinary work, including delegated and scheduled requests. */
34
34
  export function setupModeActive(input) {
35
- const blank = !(input.soul ?? "").trim() && !(input.description ?? "").trim();
36
- return blank || parseSetupCommand(input.text) !== null;
35
+ return parseSetupCommand(input.text) !== null;
37
36
  }
38
37
  // skill_manage is only ever mounted alongside the other agent tools when
39
38
  // skill authoring is turned on for this turn (OMB_SKILL_AUTHORING_ENABLED);
@@ -46,7 +45,7 @@ function folderClause(cwd) {
46
45
  : "which folder on this computer it should work in (today it has none and works in a private workspace; offer to keep that, or ask for a path)";
47
46
  }
48
47
  function buildSetupPrompt(profileAside, cwd) {
49
- return ("\n\nThis bot has not been set up yet, or the user asked you to set yourself up. Your job this conversation is to set yourself up from what the user tells you." +
48
+ return ("\n\nThe user explicitly asked you to set yourself up. For this setup request, help configure the bot from what the user tells you." +
50
49
  ` First ask at most four questions that change what you would build: what the job is, when it should happen (on demand, on a schedule, or when something arrives), which apps or accounts it touches, and ${folderClause(cwd)}.` +
51
50
  " Then, before any tool call, tell the user in plain language what you intend: who you will be, what you will do and when, where you will work, what you will need from them, and what you will not do. Wait for a yes." +
52
51
  " When they say yes, first send one message that lists the cards you are about to raise, then make the tool calls — the cards must appear after that message, never before it. After the tool calls add at most one short line and do not repeat the list." +
@@ -72,6 +72,25 @@ function redactBotAuthored(message) {
72
72
  card.summary = redactSecretsInText(card.summary);
73
73
  if (typeof card.held === "string")
74
74
  card.held = redactSecretsInText(card.held);
75
+ if (typeof card.answeredText === "string")
76
+ card.answeredText = redactSecretsInText(card.answeredText);
77
+ // Bot-authored question text sits behind the subtitle the same way a
78
+ // routine's instructions do, so it is scrubbed on the same boundary.
79
+ if (card.questionRequest) {
80
+ card.questionRequest = {
81
+ ...card.questionRequest,
82
+ questions: card.questionRequest.questions.map((question) => ({
83
+ ...question,
84
+ question: redactSecretsInText(question.question),
85
+ ...(question.header ? { header: redactSecretsInText(question.header) } : {}),
86
+ options: question.options.map((option) => ({
87
+ ...option,
88
+ label: redactSecretsInText(option.label),
89
+ ...(option.description ? { description: redactSecretsInText(option.description) } : {}),
90
+ })),
91
+ })),
92
+ };
93
+ }
75
94
  // Routine definitions are executable bot-authored text stored behind the
76
95
  // visible summary. Scrub the durable payload too so nesting it on a card
77
96
  // cannot bypass the transcript's secret-redaction boundary.
@@ -133,6 +133,14 @@ const stringOrMissing = (value) => value === undefined || typeof value === "stri
133
133
  const stringOrNullOrMissing = (value) => value === undefined || value === null || typeof value === "string";
134
134
  const numberOrNullOrMissing = (value) => value === undefined || value === null || typeof value === "number";
135
135
  const stringsOrMissing = (value) => value === undefined || (Array.isArray(value) && value.every((item) => typeof item === "string"));
136
+ /** A replayed structured ask. Only the shape the card actually reads is
137
+ * required; the rest is optional and simply absent on an older event. */
138
+ const askQuestionsOrMissing = (value) => value === undefined ||
139
+ (Array.isArray(value) &&
140
+ value.every((question) => isRecord(question) &&
141
+ typeof question.question === "string" &&
142
+ Array.isArray(question.options) &&
143
+ question.options.every((option) => isRecord(option) && typeof option.label === "string")));
136
144
  function isRuntimeEvent(value) {
137
145
  if (!isRecord(value) ||
138
146
  typeof value.eventId !== "string" ||
@@ -179,7 +187,8 @@ function isRuntimeEvent(value) {
179
187
  return ((value.requestType === "permission" || value.requestType === "question") &&
180
188
  typeof value.tool === "string" &&
181
189
  typeof value.summary === "string" &&
182
- stringsOrMissing(value.choices));
190
+ stringsOrMissing(value.choices) &&
191
+ askQuestionsOrMissing(value.questions));
183
192
  case "request.resolved":
184
193
  return ((value.behavior === "allow" || value.behavior === "deny" || value.behavior === "answer") &&
185
194
  (value.source === "user" ||
@@ -52,6 +52,19 @@ export function ensureWorkspace(botId) {
52
52
  export function workspaceDir(botId) {
53
53
  return join(WORKSPACES_DIR, botId);
54
54
  }
55
+ /** File locations, not file contents or wider tool permissions. Threads keep
56
+ * independent working directories; the same bot can find its earlier output
57
+ * without assuming that a file absent from the current directory was lost. */
58
+ export function workspaceLocationsPrompt(botId, cwd, botCwd) {
59
+ return "\n\nFile locations for this bot (absolute paths): " + JSON.stringify({
60
+ currentWorkingFolder: cwd ?? "Provider default; inspect the working directory before using relative paths",
61
+ sharedBotFolder: workspaceDir(botId),
62
+ otherThreadFiles: join(TASK_WORKSPACES_DIR, botId),
63
+ ...(botCwd ? { configuredProjectFolder: botCwd } : {}),
64
+ }) + ". Different conversations can have different working folders. For an existing file, use the exact path from the conversation; if missing here, check this bot's listed folders before saying it is gone or recreating it." +
65
+ " Follow an explicitly requested destination. Otherwise put new task output in the current working folder and report its absolute path so another thread or room can use it." +
66
+ " Do not move old files, edit another active thread's work, or read another bot's private folders without authorization. These paths do not grant additional access.";
67
+ }
55
68
  /** Lines as a person counts them: a file that ends in a newline has no
56
69
  * extra empty line after it. Every budget check and every "N lines"
57
70
  * message uses this one count. */
@@ -0,0 +1,172 @@
1
+ /**
2
+ * Structured questions raised by a provider's own "ask the human" tool —
3
+ * Claude Code's built-in `AskUserQuestion`.
4
+ *
5
+ * That tool reaches us as a PERMISSION ask (the CLI routes it through
6
+ * --permission-prompt-tool like any other tool use), which is exactly the
7
+ * wrong shape: a person cannot answer "which model should this bot use?"
8
+ * with Deny / Always allow / Allow once. So the ask is re-read here into the
9
+ * questions the model actually posed, the card renders them as choices, and
10
+ * the answer goes back as text.
11
+ *
12
+ * The payload is bot-authored, so every field is treated as untrusted: the
13
+ * shape is validated rather than cast, and the counts and lengths are capped
14
+ * so a runaway (or hostile) tool call cannot produce an unreadable card.
15
+ */
16
+ /** The provider tool whose input this module understands. */
17
+ export const ASK_USER_QUESTION_TOOL = "AskUserQuestion";
18
+ /** Caps. Claude Code's own limits are smaller (1-4 questions, 2-4 options);
19
+ * these leave room for a provider that widens them without letting a card
20
+ * grow without bound. */
21
+ export const MAX_QUESTIONS = 6;
22
+ export const MAX_OPTIONS = 12;
23
+ const MAX_QUESTION_TEXT = 400;
24
+ const MAX_LABEL = 120;
25
+ const MAX_DESCRIPTION = 400;
26
+ /** One free-text answer. Long enough for a sentence or two of context. */
27
+ export const MAX_CUSTOM_ANSWER = 2000;
28
+ function text(value, limit) {
29
+ if (typeof value !== "string")
30
+ return undefined;
31
+ const trimmed = value.trim();
32
+ return trimmed ? trimmed.slice(0, limit) : undefined;
33
+ }
34
+ function isRecord(value) {
35
+ return typeof value === "object" && value !== null && !Array.isArray(value);
36
+ }
37
+ function parseOption(value) {
38
+ // A bare string is not the documented shape, but it is the obvious
39
+ // degradation and costs one line to accept.
40
+ if (typeof value === "string") {
41
+ const label = text(value, MAX_LABEL);
42
+ return label ? { label } : null;
43
+ }
44
+ if (!isRecord(value))
45
+ return null;
46
+ const label = text(value.label, MAX_LABEL);
47
+ if (!label)
48
+ return null;
49
+ const description = text(value.description, MAX_DESCRIPTION);
50
+ return description ? { label, description } : { label };
51
+ }
52
+ function parseQuestion(value) {
53
+ if (!isRecord(value))
54
+ return null;
55
+ const question = text(value.question, MAX_QUESTION_TEXT);
56
+ if (!question)
57
+ return null;
58
+ const options = [];
59
+ const seen = new Set();
60
+ for (const raw of Array.isArray(value.options) ? value.options : []) {
61
+ const option = parseOption(raw);
62
+ // Duplicate labels are the one thing a radio group cannot survive: the
63
+ // answer text could no longer say which row was picked.
64
+ if (!option || seen.has(option.label))
65
+ continue;
66
+ seen.add(option.label);
67
+ options.push(option);
68
+ if (options.length === MAX_OPTIONS)
69
+ break;
70
+ }
71
+ // A question with nothing to choose from is still answerable — the card
72
+ // always offers free text — so an empty option list is kept, not dropped.
73
+ const header = text(value.header, MAX_LABEL);
74
+ return {
75
+ question,
76
+ ...(header ? { header } : {}),
77
+ ...(value.multiSelect === true ? { multiSelect: true } : {}),
78
+ options,
79
+ };
80
+ }
81
+ /** The questions inside an AskUserQuestion tool input, or null when the
82
+ * payload is not one (a malformed call falls back to the ordinary card). */
83
+ export function parseAskQuestions(input) {
84
+ if (!isRecord(input) || !Array.isArray(input.questions))
85
+ return null;
86
+ const questions = [];
87
+ for (const raw of input.questions) {
88
+ const question = parseQuestion(raw);
89
+ if (!question)
90
+ continue;
91
+ questions.push(question);
92
+ if (questions.length === MAX_QUESTIONS)
93
+ break;
94
+ }
95
+ return questions.length ? questions : null;
96
+ }
97
+ /** The one line the card subtitle and a spoken prompt show. */
98
+ export function askQuestionSummary(questions) {
99
+ const first = questions[0]?.question ?? "";
100
+ const rest = questions.length - 1;
101
+ return rest > 0 ? `${first} (+${rest} more question${rest > 1 ? "s" : ""})` : first;
102
+ }
103
+ /** Flat labels for clients that only know how to render a list of choices
104
+ * (the phone companions, and any older desktop build). Only a single
105
+ * question can be answered that way without losing which one was answered. */
106
+ export function questionChoices(questions) {
107
+ if (questions.length !== 1)
108
+ return undefined;
109
+ const only = questions[0];
110
+ if (only.multiSelect || only.options.length < 2)
111
+ return undefined;
112
+ return only.options.map((option) => option.label);
113
+ }
114
+ /** The lead-in on a formatted answer. It exists for the model — the answer
115
+ * is delivered on the deny channel, so it has to say what it is — and the
116
+ * card strips it back off when it shows the person what they sent. */
117
+ export const ANSWER_PREAMBLE = "The user answered your questions.";
118
+ /**
119
+ * What the model is told. It arrives as the tool's result, so it has to
120
+ * stand on its own: name each question, then what was picked for it.
121
+ */
122
+ export function formatQuestionAnswers(questions, answers) {
123
+ const blocks = [];
124
+ questions.forEach((question, index) => {
125
+ const picked = (answers[index] ?? []).map((value) => value.trim()).filter(Boolean);
126
+ if (!picked.length)
127
+ return;
128
+ blocks.push(`Q: ${question.question}\nA: ${picked.join(", ")}`);
129
+ });
130
+ if (!blocks.length)
131
+ return "";
132
+ return `${ANSWER_PREAMBLE}\n\n${blocks.join("\n\n")}`;
133
+ }
134
+ /** The same answer with the model-facing lead-in removed, for the settled
135
+ * card. Anything that does not carry the lead-in is shown as it is. */
136
+ export function answerWithoutPreamble(answer) {
137
+ return answer.startsWith(`${ANSWER_PREAMBLE}\n\n`) ? answer.slice(ANSWER_PREAMBLE.length + 2) : answer;
138
+ }
139
+ /**
140
+ * The answer text, read back as one value per question.
141
+ *
142
+ * `AskUserQuestion` is answered through its own `answers` field, keyed by the
143
+ * question's text — but the card sends ONE answer for the whole set, because
144
+ * a person answers the whole card at once. `formatQuestionAnswers` writes
145
+ * each question's text beside its answer for exactly this reason, so the map
146
+ * is recovered rather than guessed.
147
+ *
148
+ * A message that carries no blocks at all is the flat path: an older client,
149
+ * or a phone answering a single-question card with one of the option labels
150
+ * the harness also sends. With exactly one question there is no ambiguity
151
+ * about what it answers, so the whole message is that question's answer.
152
+ * With more than one there is, and nothing is filed.
153
+ */
154
+ export function questionAnswersByQuestion(message, questions) {
155
+ const answers = {};
156
+ const known = new Map(questions.map((entry) => [entry.question, entry.question]));
157
+ for (const block of message.split("\n\n")) {
158
+ const match = /^Q: ([\s\S]+?)\nA: ([\s\S]+)$/.exec(block.trim());
159
+ if (!match)
160
+ continue;
161
+ // Only a question this ask actually posed. An unrecognized block is
162
+ // dropped rather than filed under a key the tool never asked about.
163
+ const question = known.get(match[1].trim());
164
+ if (question)
165
+ answers[question] = match[2].trim();
166
+ }
167
+ if (Object.keys(answers).length)
168
+ return answers;
169
+ const only = questions.length === 1 ? questions[0] : undefined;
170
+ const flat = message.trim();
171
+ return only && flat ? { [only.question]: flat } : {};
172
+ }
@@ -0,0 +1,6 @@
1
+ /** Match provider error wording, not ordinary assistant discussions of safety. */
2
+ export function isProviderSafetyBlock(message) {
3
+ return /\bblocked by (?:our|the provider['’]s) safety systems\b|\bsafety monitoring\b.{0,100}\b(?:paused|ended|blocked)\b|\b(?:safety_check_failed|safety_policy_violation)\b/i.test(message);
4
+ }
5
+ export const PROVIDER_SAFETY_GUIDANCE = "The provider stopped this task. Full access controls tool approvals, not provider safety checks. Review the provider’s findings in Codex if available; OpenMausBot cannot override this block.";
6
+ export const PROVIDER_SAFETY_HELP_URL = "https://learn.chatgpt.com/docs/agent-approvals-security#safety-monitoring-and-paused-tasks";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmausbot",
3
- "version": "0.1.74",
3
+ "version": "0.1.75",
4
4
  "description": "Run the OpenMausBot server anywhere and pair your devices to it",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",