llm_meta_widget 0.4.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: e6113516f03758b2dc602dd4f2e62b515206821b710c646c07faf474e1f9aa48
|
|
4
|
+
data.tar.gz: 5d65ceb25848ba253dfc61f92f04ed6f1110581fa5c3b7640c26718f2c4c4e51
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 6b255c67fa8f44ccc1a7fcb71a2ccb149bf1e02f1441c7bbda31a22502f98c491902dd978c7817e411af5865a54557a89581ec36dbd64e41ada6a1722397d04d
|
|
7
|
+
data.tar.gz: dd5fa208a4fbb758fdcf7b9683f6102ea49b99270fd5a0e49182ab7eb21966705b6aca5ae7ee746694dd3a019ef6bdf662d0d24f57551dd2a8b86e0921cee0cd
|
data/README.md
CHANGED
|
@@ -8,19 +8,25 @@ Client-orchestrated: the widget fetches host-side action schemas + host-publishe
|
|
|
8
8
|
|
|
9
9
|
## What you need first
|
|
10
10
|
|
|
11
|
-
The widget is a browser front end
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
11
|
+
The widget is a browser front end: it needs something to answer the chat. That
|
|
12
|
+
can be **[llm_meta_server](https://github.com/pubannotation)** — a hub that
|
|
13
|
+
holds provider credentials and streams responses back — or **an Ollama**,
|
|
14
|
+
which the widget talks to directly with no server component of ours in
|
|
15
|
+
between.
|
|
15
16
|
|
|
16
|
-
|
|
17
|
-
|
|
17
|
+
Those are separate from where tools come from. A hub also registers MCP tools
|
|
18
|
+
(Class 1 below), and you can point at one for tools while a local Ollama
|
|
19
|
+
answers the chat, or use no hub at all. Page actions and your own
|
|
20
|
+
`.well-known/mcp.json` never involve a hub either way.
|
|
21
|
+
|
|
22
|
+
Nothing else is required: no database, no migrations, no JavaScript build
|
|
23
|
+
step, no Node at runtime.
|
|
18
24
|
|
|
19
25
|
| Requirement | Version |
|
|
20
26
|
|---|---|
|
|
21
27
|
| Ruby | >= 3.2 |
|
|
22
28
|
| Rails | >= 8.0 (8.1 not required) |
|
|
23
|
-
|
|
|
29
|
+
| An llm_meta_server **or** an Ollama | any |
|
|
24
30
|
|
|
25
31
|
## Installation
|
|
26
32
|
|
|
@@ -45,7 +51,7 @@ Put this on any view — a fresh `pages/demo.html.erb` is fine:
|
|
|
45
51
|
```erb
|
|
46
52
|
<h1>Demo</h1>
|
|
47
53
|
|
|
48
|
-
<%= llm_meta_widget(
|
|
54
|
+
<%= llm_meta_widget(llm_url: "https://your-meta-server.example",
|
|
49
55
|
model: "qwen3-6-35b-fast",
|
|
50
56
|
greeting: "Hi — ask me anything about this page.") %>
|
|
51
57
|
```
|
|
@@ -63,8 +69,9 @@ says so explicitly.
|
|
|
63
69
|
|
|
64
70
|
Two more things worth knowing before you go further:
|
|
65
71
|
|
|
66
|
-
- `
|
|
67
|
-
your server. `localhost` works only while you are the visitor
|
|
72
|
+
- `llm_url` must be reachable from your visitors' browsers, not just from
|
|
73
|
+
your server. `localhost` works only while you are the visitor — which is
|
|
74
|
+
fine when each visitor runs their own Ollama.
|
|
68
75
|
- The model name is the hub's name for it (`GET /api/llms` lists them), not
|
|
69
76
|
the provider's.
|
|
70
77
|
|
|
@@ -185,7 +192,8 @@ To adjust picker behavior at the helper call site:
|
|
|
185
192
|
|
|
186
193
|
```erb
|
|
187
194
|
<%= llm_meta_widget(
|
|
188
|
-
|
|
195
|
+
llm_url: "https://your-meta-server.example",
|
|
196
|
+
tool_hub_url: "https://your-meta-server.example",
|
|
189
197
|
model: "qwen3-6-35b-fast", # initial selection
|
|
190
198
|
enable_model_picker: true, # false → hide picker, use fixed `model:`
|
|
191
199
|
enable_tool_picker: true, # false → hide picker, no Class-1 tools
|
|
@@ -197,7 +205,7 @@ To adjust picker behavior at the helper call site:
|
|
|
197
205
|
To lock the widget to the fixed `model:` prop and disable Class-1 tools entirely (level-0 mode — Class 2 & 3 still work):
|
|
198
206
|
|
|
199
207
|
```erb
|
|
200
|
-
<%= llm_meta_widget(
|
|
208
|
+
<%= llm_meta_widget(llm_url: "…", model: "…",
|
|
201
209
|
enable_model_picker: false,
|
|
202
210
|
enable_tool_picker: false) %>
|
|
203
211
|
```
|
|
@@ -206,7 +214,9 @@ To lock the widget to the fixed `model:` prop and disable Class-1 tools entirely
|
|
|
206
214
|
|
|
207
215
|
| Option | Default | Purpose |
|
|
208
216
|
|---|---|---|
|
|
209
|
-
| `
|
|
217
|
+
| `llm_url:` | required | Who answers the chat — an llm_meta_server, or an Ollama |
|
|
218
|
+
| `llm_provider:` | `:llm_meta_server` | What `llm_url` speaks. `:ollama` talks to Ollama's `/api/chat` directly |
|
|
219
|
+
| `tool_hub_url:` | `nil` | An llm_meta_server whose registered MCP tools to offer. Absent = none (page actions and your own `.well-known` MCP are unaffected) |
|
|
210
220
|
| `model:` | required | Initial model (also the fallback when picker is disabled) |
|
|
211
221
|
| `api_key_uuid:` | `"ollama-local"` | Hub API-key uuid to invoke |
|
|
212
222
|
| `orchestrator_path:` | `"/llm_meta_widget_assets/orchestrator.js"` | Served by the gem's engine; rarely overridden |
|
|
@@ -222,6 +232,45 @@ To lock the widget to the fixed `model:` prop and disable Class-1 tools entirely
|
|
|
222
232
|
| `models:` | `nil` | Model-name allowlist; `nil` = all anon-available |
|
|
223
233
|
| `hub_tools:` | `nil` | MCP-server-name allowlist; `nil` = all anon-public |
|
|
224
234
|
|
|
235
|
+
## Choosing what answers the chat, and whose tools to offer
|
|
236
|
+
|
|
237
|
+
Two independent settings, because they are two jobs. Neither implies the
|
|
238
|
+
other, so there is nothing to disable and nothing to inherit — what a widget
|
|
239
|
+
does is what its call site says.
|
|
240
|
+
|
|
241
|
+
```erb
|
|
242
|
+
<%# a hub answers the chat; its registered tools are offered too %>
|
|
243
|
+
<% hub = "https://your-meta-server.example" %>
|
|
244
|
+
<%= llm_meta_widget(llm_url: hub, tool_hub_url: hub, model: "qwen3-6-35b-fast") %>
|
|
245
|
+
|
|
246
|
+
<%# a hub answers the chat; no hub-registered tools %>
|
|
247
|
+
<%= llm_meta_widget(llm_url: hub, model: "qwen3-6-35b-fast") %>
|
|
248
|
+
|
|
249
|
+
<%# the visitor's own Ollama answers; a hub still supplies its tools %>
|
|
250
|
+
<%= llm_meta_widget(llm_url: "http://localhost:11434", llm_provider: :ollama,
|
|
251
|
+
tool_hub_url: hub, model: "qwen3.8:27b") %>
|
|
252
|
+
|
|
253
|
+
<%# nothing of ours in the path at all %>
|
|
254
|
+
<%= llm_meta_widget(llm_url: "http://localhost:11434", llm_provider: :ollama,
|
|
255
|
+
model: "qwen3.8:27b") %>
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
**The Ollama path.** The widget POSTs to `{llm_url}/api/chat` and reads
|
|
259
|
+
Ollama's NDJSON stream; tool schemas go over as OpenAI-shaped functions and
|
|
260
|
+
`tool_calls` come back, so Class 2 and Class 3 tools work exactly as they do
|
|
261
|
+
against a hub. The model picker lists `{llm_url}/api/tags`. Ollama must be
|
|
262
|
+
told to accept your page's origin — `OLLAMA_ORIGINS=https://your-site.example`
|
|
263
|
+
— which is the same CORS story as the hub, configured elsewhere.
|
|
264
|
+
|
|
265
|
+
What you give up without a `tool_hub_url` is Class 1 only: tools registered on
|
|
266
|
+
somebody's hub. That is not "no tools" — page actions and your own
|
|
267
|
+
`.well-known/mcp.json` are untouched, and they are the interesting ones for an
|
|
268
|
+
assistant embedded in your page.
|
|
269
|
+
|
|
270
|
+
**Why anyone would want the last shape:** with a local Ollama and local MCP
|
|
271
|
+
endpoints, nothing a visitor types leaves the machine. No credentials to hold,
|
|
272
|
+
no retention policy to write.
|
|
273
|
+
|
|
225
274
|
## Declaring resources and prompts (static-primitives extension)
|
|
226
275
|
|
|
227
276
|
If you operate the MCP endpoint behind your `.well-known/mcp.json`, you can
|
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
// onTextDelta: (str) => {},
|
|
33
33
|
// onThinkingDelta: (str) => {},
|
|
34
34
|
// onToolCall: (tc) => {}, // { id, name, arguments }
|
|
35
|
+
// onToolDispatched: ({toolCall, value, error}) => {}, // after it ran
|
|
35
36
|
// onPhase: (name) => {}, // 'thinking' | 'tool_execution' | ...
|
|
36
37
|
// signal: abortController.signal
|
|
37
38
|
// })
|
|
@@ -324,11 +325,24 @@ export async function runChatLoop(opts) {
|
|
|
324
325
|
remoteTools = [],
|
|
325
326
|
hostWideTools = [],
|
|
326
327
|
maxRounds = 10,
|
|
328
|
+
// "hub" (default) talks to llm_meta_server; "ollama" talks straight to an
|
|
329
|
+
// Ollama, with no server component of ours in between. The loop is the
|
|
330
|
+
// same either way — only the call is swapped.
|
|
331
|
+
provider = "hub",
|
|
332
|
+
// Where Class 1 (hub-registered) tools are proxied. Usually the same
|
|
333
|
+
// llm_meta_server that answers the chat, but not when the chat goes
|
|
334
|
+
// straight to an Ollama — the tools still belong to the hub.
|
|
335
|
+
toolHubUrl,
|
|
327
336
|
signal,
|
|
328
337
|
onRoundStart,
|
|
329
338
|
onTextDelta,
|
|
330
339
|
onThinkingDelta,
|
|
331
340
|
onToolCall,
|
|
341
|
+
// Fired as each tool finishes, so a caller can show what ran WHILE the
|
|
342
|
+
// turn is still going. The returned `dispatched` array says the same
|
|
343
|
+
// thing, but only once the whole loop ends — which on a slow model is
|
|
344
|
+
// minutes after the work happened, with the user watching unnamed tools.
|
|
345
|
+
onToolDispatched,
|
|
332
346
|
onPhase,
|
|
333
347
|
...singleOpts
|
|
334
348
|
} = opts
|
|
@@ -393,7 +407,9 @@ export async function runChatLoop(opts) {
|
|
|
393
407
|
if (signal?.aborted) throw new DOMException("aborted", "AbortError")
|
|
394
408
|
onRoundStart?.(round)
|
|
395
409
|
|
|
396
|
-
|
|
410
|
+
// The loop is the same whoever answers; only the call differs.
|
|
411
|
+
const call = provider === "ollama" ? ollamaChatCall : singleLlmCall
|
|
412
|
+
const turnResult = await call({
|
|
397
413
|
...singleOpts,
|
|
398
414
|
messages,
|
|
399
415
|
toolIds,
|
|
@@ -454,7 +470,8 @@ export async function runChatLoop(opts) {
|
|
|
454
470
|
// the write. It also means a failed action is something the model can
|
|
455
471
|
// see and correct, instead of a red mark only the user notices.
|
|
456
472
|
const roundTripResults = []
|
|
457
|
-
for (const { toolCall, error } of localOut.dispatched) {
|
|
473
|
+
for (const { toolCall, value, error } of localOut.dispatched) {
|
|
474
|
+
onToolDispatched?.({ toolCall, value, error })
|
|
458
475
|
roundTripResults.push({
|
|
459
476
|
tc: toolCall,
|
|
460
477
|
result: error ? { error: String(error.message || error) } : { ok: true, applied: toolCall.name }
|
|
@@ -468,9 +485,11 @@ export async function runChatLoop(opts) {
|
|
|
468
485
|
try {
|
|
469
486
|
const value = await callMcpTool({ endpoint: tool.endpoint, name: tc.name, args, signal })
|
|
470
487
|
allDispatched.push({ toolCall: tc, value })
|
|
488
|
+
onToolDispatched?.({ toolCall: tc, value })
|
|
471
489
|
roundTripResults.push({ tc, result: value })
|
|
472
490
|
} catch (error) {
|
|
473
491
|
allDispatched.push({ toolCall: tc, error })
|
|
492
|
+
onToolDispatched?.({ toolCall: tc, error })
|
|
474
493
|
roundTripResults.push({ tc, result: { error: String(error.message || error) } })
|
|
475
494
|
}
|
|
476
495
|
}
|
|
@@ -481,16 +500,18 @@ export async function runChatLoop(opts) {
|
|
|
481
500
|
const args = coerceArguments(tc.arguments)
|
|
482
501
|
try {
|
|
483
502
|
const value = await dispatchRemoteToolCall({
|
|
484
|
-
baseUrl: singleOpts.baseUrl,
|
|
503
|
+
baseUrl: toolHubUrl || singleOpts.baseUrl,
|
|
485
504
|
bearerToken: singleOpts.bearerToken,
|
|
486
505
|
toolId: tool.id,
|
|
487
506
|
args: args,
|
|
488
507
|
signal
|
|
489
508
|
})
|
|
490
509
|
allDispatched.push({ toolCall: tc, value })
|
|
510
|
+
onToolDispatched?.({ toolCall: tc, value })
|
|
491
511
|
roundTripResults.push({ tc, result: value })
|
|
492
512
|
} catch (error) {
|
|
493
513
|
allDispatched.push({ toolCall: tc, error })
|
|
514
|
+
onToolDispatched?.({ toolCall: tc, error })
|
|
494
515
|
// Feed the error text back to the LLM as the tool result — better
|
|
495
516
|
// than dropping it (the LLM can react, apologize, retry differently).
|
|
496
517
|
roundTripResults.push({ tc, result: { error: String(error.message || error) } })
|
|
@@ -995,3 +1016,172 @@ export function promptButtonProps(prompt) {
|
|
|
995
1016
|
}
|
|
996
1017
|
}
|
|
997
1018
|
|
|
1019
|
+
|
|
1020
|
+
// ---- direct-to-Ollama provider ------------------------------------------
|
|
1021
|
+
//
|
|
1022
|
+
// The smallest adoption: a Rails app and an Ollama, with no hub of ours in
|
|
1023
|
+
// between. Ollama's /api/chat already does the hard part — it takes tool
|
|
1024
|
+
// schemas and returns tool_calls — so what differs from the hub path is the
|
|
1025
|
+
// envelope: NDJSON rather than SSE, and its own request shape.
|
|
1026
|
+
//
|
|
1027
|
+
// What an adopter gives up is Class 1 (hub-registered MCP), which is the
|
|
1028
|
+
// hub's job by definition. Page actions and the host's own .well-known MCP
|
|
1029
|
+
// both work unchanged, and nothing the visitor types leaves their machine.
|
|
1030
|
+
//
|
|
1031
|
+
// Ollama must be told to accept the page's origin (OLLAMA_ORIGINS), the same
|
|
1032
|
+
// CORS story as the hub, configured elsewhere.
|
|
1033
|
+
|
|
1034
|
+
export async function* parseNdjsonStream(readableStream, signal) {
|
|
1035
|
+
const reader = readableStream.getReader()
|
|
1036
|
+
const decoder = new TextDecoder("utf-8")
|
|
1037
|
+
let buffer = ""
|
|
1038
|
+
|
|
1039
|
+
const onAbort = () => { try { reader.cancel() } catch { /* noop */ } }
|
|
1040
|
+
signal?.addEventListener("abort", onAbort)
|
|
1041
|
+
|
|
1042
|
+
try {
|
|
1043
|
+
while (true) {
|
|
1044
|
+
const { value, done } = await reader.read()
|
|
1045
|
+
if (done) break
|
|
1046
|
+
buffer += decoder.decode(value, { stream: true })
|
|
1047
|
+
|
|
1048
|
+
let nl
|
|
1049
|
+
while ((nl = buffer.indexOf("\n")) !== -1) {
|
|
1050
|
+
const line = buffer.slice(0, nl).trim()
|
|
1051
|
+
buffer = buffer.slice(nl + 1)
|
|
1052
|
+
if (!line) continue
|
|
1053
|
+
try {
|
|
1054
|
+
yield JSON.parse(line)
|
|
1055
|
+
} catch (e) {
|
|
1056
|
+
// A truncated or non-JSON line is not worth aborting a stream for.
|
|
1057
|
+
}
|
|
1058
|
+
}
|
|
1059
|
+
}
|
|
1060
|
+
const tail = buffer.trim()
|
|
1061
|
+
if (tail) { try { yield JSON.parse(tail) } catch (e) { /* ignore */ } }
|
|
1062
|
+
} finally {
|
|
1063
|
+
signal?.removeEventListener("abort", onAbort)
|
|
1064
|
+
try { reader.releaseLock() } catch { /* noop */ }
|
|
1065
|
+
}
|
|
1066
|
+
}
|
|
1067
|
+
|
|
1068
|
+
// Tool schemas travel as MCP-flavoured `input_schema`; Ollama wants OpenAI's
|
|
1069
|
+
// function shape.
|
|
1070
|
+
function toolsForOllama(localTools) {
|
|
1071
|
+
return (localTools || []).map((t) => ({
|
|
1072
|
+
type: "function",
|
|
1073
|
+
function: {
|
|
1074
|
+
name: t.name,
|
|
1075
|
+
description: t.description,
|
|
1076
|
+
parameters: t.input_schema || t.inputSchema || { type: "object", properties: {} }
|
|
1077
|
+
}
|
|
1078
|
+
}))
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
// The hub's history carries `tool_call_id`; Ollama identifies a result by the
|
|
1082
|
+
// tool's name instead, and rejects unknown keys on some versions.
|
|
1083
|
+
function messagesForOllama(messages) {
|
|
1084
|
+
return (messages || []).map((m) => {
|
|
1085
|
+
if (m.role === "tool") {
|
|
1086
|
+
return { role: "tool", tool_name: m.name, content: m.content }
|
|
1087
|
+
}
|
|
1088
|
+
if (m.tool_calls) {
|
|
1089
|
+
return {
|
|
1090
|
+
role: m.role,
|
|
1091
|
+
content: m.content || "",
|
|
1092
|
+
tool_calls: m.tool_calls.map((tc) => ({
|
|
1093
|
+
function: { name: tc.name, arguments: coerceArguments(tc.arguments) }
|
|
1094
|
+
}))
|
|
1095
|
+
}
|
|
1096
|
+
}
|
|
1097
|
+
return { role: m.role, content: m.content }
|
|
1098
|
+
})
|
|
1099
|
+
}
|
|
1100
|
+
|
|
1101
|
+
// Same signature and return shape as singleLlmCall, so runChatLoop does not
|
|
1102
|
+
// care which provider answered.
|
|
1103
|
+
export async function ollamaChatCall({
|
|
1104
|
+
baseUrl,
|
|
1105
|
+
modelName,
|
|
1106
|
+
messages,
|
|
1107
|
+
localTools = [],
|
|
1108
|
+
generationSettings = {},
|
|
1109
|
+
onTextDelta,
|
|
1110
|
+
onThinkingDelta,
|
|
1111
|
+
onToolCall,
|
|
1112
|
+
onPhase,
|
|
1113
|
+
signal,
|
|
1114
|
+
}) {
|
|
1115
|
+
if (!baseUrl || !modelName) throw new Error("ollamaChatCall: baseUrl and modelName are required")
|
|
1116
|
+
if (!Array.isArray(messages) || messages.length === 0) {
|
|
1117
|
+
throw new Error("ollamaChatCall: messages must be a non-empty array")
|
|
1118
|
+
}
|
|
1119
|
+
|
|
1120
|
+
const { think, ...options } = generationSettings || {}
|
|
1121
|
+
const body = { model: modelName, messages: messagesForOllama(messages), stream: true }
|
|
1122
|
+
if (localTools && localTools.length) body.tools = toolsForOllama(localTools)
|
|
1123
|
+
if (think !== undefined) body.think = think
|
|
1124
|
+
if (Object.keys(options).length) body.options = options
|
|
1125
|
+
|
|
1126
|
+
const response = await fetch(`${baseUrl.replace(/\/$/, "")}/api/chat`, {
|
|
1127
|
+
method: "POST",
|
|
1128
|
+
headers: { "Content-Type": "application/json" },
|
|
1129
|
+
body: JSON.stringify(body),
|
|
1130
|
+
signal
|
|
1131
|
+
})
|
|
1132
|
+
if (!response.ok) {
|
|
1133
|
+
const text = await response.text().catch(() => "")
|
|
1134
|
+
throw new Error(`ollamaChatCall: HTTP ${response.status} ${response.statusText}${text ? " — " + text.slice(0, 200) : ""}`)
|
|
1135
|
+
}
|
|
1136
|
+
if (!response.body) throw new Error("ollamaChatCall: response has no body (streaming unsupported?)")
|
|
1137
|
+
|
|
1138
|
+
const toolCalls = []
|
|
1139
|
+
let content = ""
|
|
1140
|
+
let finishReason = null
|
|
1141
|
+
let announcedPhase = null
|
|
1142
|
+
|
|
1143
|
+
const phase = (name) => {
|
|
1144
|
+
if (announcedPhase === name) return
|
|
1145
|
+
announcedPhase = name
|
|
1146
|
+
onPhase?.(name)
|
|
1147
|
+
}
|
|
1148
|
+
|
|
1149
|
+
for await (const frame of parseNdjsonStream(response.body, signal)) {
|
|
1150
|
+
const message = frame.message || {}
|
|
1151
|
+
|
|
1152
|
+
if (message.thinking) {
|
|
1153
|
+
phase("thinking")
|
|
1154
|
+
onThinkingDelta?.(message.thinking)
|
|
1155
|
+
}
|
|
1156
|
+
if (message.content) {
|
|
1157
|
+
phase("responding")
|
|
1158
|
+
content += message.content
|
|
1159
|
+
onTextDelta?.(message.content)
|
|
1160
|
+
}
|
|
1161
|
+
for (const tc of message.tool_calls || []) {
|
|
1162
|
+
const call = {
|
|
1163
|
+
id: tc.id || `ollama-${toolCalls.length}`,
|
|
1164
|
+
name: tc.function?.name,
|
|
1165
|
+
arguments: tc.function?.arguments ?? {}
|
|
1166
|
+
}
|
|
1167
|
+
toolCalls.push(call)
|
|
1168
|
+
onToolCall?.(call)
|
|
1169
|
+
}
|
|
1170
|
+
if (frame.done) finishReason = frame.done_reason || "stop"
|
|
1171
|
+
}
|
|
1172
|
+
|
|
1173
|
+
return { content, toolCalls, finishReason }
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
// Ollama's own catalogue, for the model picker when there is no hub to ask.
|
|
1177
|
+
export async function fetchOllamaModels({ baseUrl, signal }) {
|
|
1178
|
+
try {
|
|
1179
|
+
const response = await fetch(`${baseUrl.replace(/\/$/, "")}/api/tags`, { signal })
|
|
1180
|
+
if (!response.ok) return []
|
|
1181
|
+
const data = await response.json()
|
|
1182
|
+
return (data.models || []).map((m) => m.name).filter(Boolean)
|
|
1183
|
+
} catch (e) {
|
|
1184
|
+
// A picker that cannot be populated is not a reason to break the widget.
|
|
1185
|
+
return []
|
|
1186
|
+
}
|
|
1187
|
+
}
|
|
@@ -16,7 +16,18 @@ module LlmMetaWidget
|
|
|
16
16
|
# (page-embedded aiActions / host-wide well-known / hub-registered).
|
|
17
17
|
module WidgetHelper
|
|
18
18
|
DEFAULTS = {
|
|
19
|
-
api_key_uuid: "ollama-local",
|
|
19
|
+
api_key_uuid: "ollama-local", # llm_meta_server provider only
|
|
20
|
+
# Answering the chat and registering tools are separate jobs. Name each
|
|
21
|
+
# endpoint for the job it does; neither implies the other.
|
|
22
|
+
#
|
|
23
|
+
# llm_url: who answers the chat (required)
|
|
24
|
+
# llm_provider: what it speaks — :llm_meta_server (default) or :ollama
|
|
25
|
+
# tool_hub_url: an llm_meta_server whose registered MCP tools to
|
|
26
|
+
# offer. Absent means none — which does NOT mean "no
|
|
27
|
+
# tools": page actions and the host's own
|
|
28
|
+
# .well-known/mcp.json are unaffected.
|
|
29
|
+
llm_provider: "llm_meta_server",
|
|
30
|
+
tool_hub_url: nil,
|
|
20
31
|
orchestrator_path: "/llm_meta_widget_assets/orchestrator.js",
|
|
21
32
|
actions_schema_id: "ai-actions",
|
|
22
33
|
state_global: "aiState",
|
|
@@ -42,9 +53,20 @@ module LlmMetaWidget
|
|
|
42
53
|
greeting: nil
|
|
43
54
|
}.freeze
|
|
44
55
|
|
|
45
|
-
def llm_meta_widget(
|
|
46
|
-
locals = DEFAULTS.merge(
|
|
47
|
-
|
|
56
|
+
def llm_meta_widget(llm_url:, model:, **overrides)
|
|
57
|
+
locals = DEFAULTS.merge(llm_url: llm_url, model: model, **overrides)
|
|
58
|
+
provider = locals[:llm_provider].to_s
|
|
59
|
+
unless %w[llm_meta_server ollama].include?(provider)
|
|
60
|
+
raise ArgumentError,
|
|
61
|
+
"llm_meta_widget: llm_provider must be :llm_meta_server or :ollama, got #{provider.inspect}"
|
|
62
|
+
end
|
|
63
|
+
raise ArgumentError, "llm_meta_widget: llm_url is required" if llm_url.to_s.strip.empty?
|
|
64
|
+
|
|
65
|
+
hub = locals[:tool_hub_url]
|
|
66
|
+
hub = nil if hub.to_s.strip.empty?
|
|
67
|
+
|
|
68
|
+
render partial: "llm_meta_widget/chat_panel",
|
|
69
|
+
locals: locals.merge(llm_provider: provider, tool_hub_url: hub)
|
|
48
70
|
end
|
|
49
71
|
end
|
|
50
72
|
end
|
|
@@ -452,7 +452,7 @@
|
|
|
452
452
|
import { runChatLoop, fetchMcpManifest, listMcpPrompts, getMcpPrompt,
|
|
453
453
|
listMcpResources, readMcpResource, promptMessagesToText,
|
|
454
454
|
loadHostResource, resourceLinesForTurn, resolvePromptArguments,
|
|
455
|
-
promptButtonProps, promptArgumentSummary } from "<%= orchestrator_path %>";
|
|
455
|
+
promptButtonProps, promptArgumentSummary, fetchOllamaModels } from "<%= orchestrator_path %>";
|
|
456
456
|
import { marked } from "/llm_meta_widget_assets/marked.esm.js";
|
|
457
457
|
|
|
458
458
|
// Standard prose settings — GFM (tables, autolinks, strikethrough),
|
|
@@ -460,7 +460,12 @@
|
|
|
460
460
|
// bare newlines still reads naturally.
|
|
461
461
|
marked.setOptions({ gfm: true, breaks: true });
|
|
462
462
|
|
|
463
|
-
|
|
463
|
+
// Two endpoints, two jobs: one answers the chat, the other registers the
|
|
464
|
+
// MCP tools on offer. They are usually the same llm_meta_server, but need
|
|
465
|
+
// not be — an adopter can run their own Ollama and still borrow a hub's
|
|
466
|
+
// tools, or use no hub at all.
|
|
467
|
+
var LLM_BASE = <%= llm_url.to_json.html_safe %>;
|
|
468
|
+
var TOOL_HUB_BASE = <%= tool_hub_url.to_json.html_safe %>;
|
|
464
469
|
var API_KEY_UUID = <%= api_key_uuid.to_json.html_safe %>;
|
|
465
470
|
var MODEL = <%= model.to_json.html_safe %>;
|
|
466
471
|
var ACTIONS_SCHEMA_ID = <%= actions_schema_id.to_json.html_safe %>;
|
|
@@ -470,8 +475,12 @@
|
|
|
470
475
|
var REMOTE_TOOLS_SCHEMA_ID = <%= remote_tools_schema_id.to_json.html_safe %>;
|
|
471
476
|
var MAX_ROUNDS = <%= max_rounds.to_json.html_safe %>;
|
|
472
477
|
var WELL_KNOWN_URLS = <%= raw(well_known_urls.nil? ? "null" : well_known_urls.to_json) %>;
|
|
478
|
+
var LLM_PROVIDER = <%= llm_provider.to_json.html_safe %>;
|
|
473
479
|
var ENABLE_MODEL_PICKER = <%= enable_model_picker.to_json.html_safe %>;
|
|
474
|
-
|
|
480
|
+
// Class 1 tools are registered on a hub, so the picker needs one — which
|
|
481
|
+
// is independent of who answers the chat. Page actions and the host's own
|
|
482
|
+
// .well-known MCP work regardless.
|
|
483
|
+
var ENABLE_TOOL_PICKER = <%= enable_tool_picker.to_json.html_safe %> && <%= tool_hub_url.to_json.html_safe %> !== null;
|
|
475
484
|
var MODEL_ALLOWLIST = <%= raw(models.nil? ? "null" : models.to_json) %>;
|
|
476
485
|
var HUB_TOOLS_ALLOWLIST = <%= raw(hub_tools.nil? ? "null" : hub_tools.to_json) %>;
|
|
477
486
|
|
|
@@ -633,8 +642,27 @@
|
|
|
633
642
|
// still works with the initial `model:` + no hub tools).
|
|
634
643
|
var tasks = [];
|
|
635
644
|
|
|
636
|
-
if (ENABLE_MODEL_PICKER && modelPicker) {
|
|
637
|
-
|
|
645
|
+
if (ENABLE_MODEL_PICKER && modelPicker && LLM_PROVIDER === "ollama") {
|
|
646
|
+
// No hub to ask: Ollama lists its own models.
|
|
647
|
+
tasks.push(fetchOllamaModels({ baseUrl: LLM_BASE }).then(function(names) {
|
|
648
|
+
var flat = names
|
|
649
|
+
.filter(function(n) { return anyAllowedByAllowlist(n, MODEL_ALLOWLIST); })
|
|
650
|
+
.map(function(n) { return { value: n, label: n }; });
|
|
651
|
+
if (!flat.some(function(m) { return m.value === MODEL; })) {
|
|
652
|
+
flat.unshift({ value: MODEL, label: MODEL });
|
|
653
|
+
}
|
|
654
|
+
modelPicker.innerHTML = "";
|
|
655
|
+
flat.forEach(function(m) {
|
|
656
|
+
var opt = document.createElement("option");
|
|
657
|
+
opt.value = m.value;
|
|
658
|
+
opt.textContent = m.label;
|
|
659
|
+
if (m.value === MODEL) opt.selected = true;
|
|
660
|
+
modelPicker.appendChild(opt);
|
|
661
|
+
});
|
|
662
|
+
if (flat.length > 1) modelPicker.style.display = "";
|
|
663
|
+
}));
|
|
664
|
+
} else if (ENABLE_MODEL_PICKER && modelPicker) {
|
|
665
|
+
tasks.push(fetch(LLM_BASE + "/api/llms", { headers: { "Accept": "application/json" } })
|
|
638
666
|
.then(function(r) { return r.ok ? r.json() : { llms: [] }; })
|
|
639
667
|
.then(function(payload) {
|
|
640
668
|
// /api/llms is heterogeneous per family:
|
|
@@ -671,7 +699,7 @@
|
|
|
671
699
|
}
|
|
672
700
|
|
|
673
701
|
if (ENABLE_TOOL_PICKER && toolsListEl) {
|
|
674
|
-
tasks.push(fetch(
|
|
702
|
+
tasks.push(fetch(TOOL_HUB_BASE + "/api/mcp_servers", { headers: { "Accept": "application/json" } })
|
|
675
703
|
.then(function(r) { return r.ok ? r.json() : { mcp_servers: [] }; })
|
|
676
704
|
.then(function(payload) {
|
|
677
705
|
hubMcpServers = (payload.mcp_servers || []).filter(function(s) {
|
|
@@ -1070,6 +1098,48 @@
|
|
|
1070
1098
|
label.textContent = roleLabel("assistant");
|
|
1071
1099
|
}
|
|
1072
1100
|
|
|
1101
|
+
// Tool chips, written as the tools run rather than assembled at the end.
|
|
1102
|
+
// The end-of-turn footer was invisible for as long as the turn lasted —
|
|
1103
|
+
// minutes on a thinking model — so a visitor watched tools execute with
|
|
1104
|
+
// no idea which ones.
|
|
1105
|
+
var liveChips = null; // the container inside the current bubble
|
|
1106
|
+
var pendingChips = []; // chips awaiting their outcome, oldest first
|
|
1107
|
+
|
|
1108
|
+
function chipRow() {
|
|
1109
|
+
if (liveChips && liveChips.parentNode) return liveChips;
|
|
1110
|
+
if (!activeAssistantBody || !activeAssistantBody.parentNode) return null;
|
|
1111
|
+
liveChips = document.createElement("div");
|
|
1112
|
+
liveChips.className = "lmw-tool-chips";
|
|
1113
|
+
// Inside the bubble but OUTSIDE .message-content: every text delta
|
|
1114
|
+
// re-renders that element's markdown from scratch, which silently
|
|
1115
|
+
// erased any chip already written into it.
|
|
1116
|
+
activeAssistantBody.parentNode.appendChild(liveChips);
|
|
1117
|
+
return liveChips;
|
|
1118
|
+
}
|
|
1119
|
+
|
|
1120
|
+
function announceToolCall(toolCall) {
|
|
1121
|
+
var row = chipRow();
|
|
1122
|
+
if (!row) return;
|
|
1123
|
+
var chip = document.createElement("span");
|
|
1124
|
+
chip.className = "lmw-tool-chip running";
|
|
1125
|
+
chip.textContent = "⏳ " + toolCall.name;
|
|
1126
|
+
chip.dataset.tool = toolCall.name;
|
|
1127
|
+
row.appendChild(chip);
|
|
1128
|
+
pendingChips.push(chip);
|
|
1129
|
+
historyEl.scrollTop = historyEl.scrollHeight;
|
|
1130
|
+
}
|
|
1131
|
+
|
|
1132
|
+
function resolveToolCall(outcome) {
|
|
1133
|
+
// Match by name, oldest first: ids are not always echoed back, and a
|
|
1134
|
+
// repeated name resolves in call order.
|
|
1135
|
+
var idx = pendingChips.findIndex(function(c) { return c.dataset.tool === outcome.toolCall.name; });
|
|
1136
|
+
var chip = idx === -1 ? null : pendingChips.splice(idx, 1)[0];
|
|
1137
|
+
if (!chip) return;
|
|
1138
|
+
chip.className = "lmw-tool-chip" + (outcome.error ? " error" : "");
|
|
1139
|
+
chip.textContent = (outcome.error ? "❌ " : "🔧 ") + outcome.toolCall.name;
|
|
1140
|
+
if (outcome.error) chip.title = outcome.error.message || String(outcome.error);
|
|
1141
|
+
}
|
|
1142
|
+
|
|
1073
1143
|
var currentThinkingBlock = null;
|
|
1074
1144
|
var currentThinkingBody = null;
|
|
1075
1145
|
// The assistant bubble of the turn in flight. Reasoning belongs ABOVE that
|
|
@@ -1196,6 +1266,8 @@
|
|
|
1196
1266
|
|
|
1197
1267
|
var assistantBody = appendTurn("assistant", "");
|
|
1198
1268
|
activeAssistantBody = assistantBody;
|
|
1269
|
+
liveChips = null;
|
|
1270
|
+
pendingChips = [];
|
|
1199
1271
|
markWorking(assistantBody.roleLabel);
|
|
1200
1272
|
var assistantMarkdown = ""; // accumulate raw markdown, re-render on each delta
|
|
1201
1273
|
|
|
@@ -1212,7 +1284,8 @@
|
|
|
1212
1284
|
.concat([{ role: "user", content: userText }]);
|
|
1213
1285
|
|
|
1214
1286
|
var result = await runChatLoop({
|
|
1215
|
-
baseUrl:
|
|
1287
|
+
baseUrl: LLM_BASE,
|
|
1288
|
+
toolHubUrl: TOOL_HUB_BASE,
|
|
1216
1289
|
apiKeyUuid: API_KEY_UUID,
|
|
1217
1290
|
modelName: MODEL,
|
|
1218
1291
|
messages: messages,
|
|
@@ -1221,7 +1294,10 @@
|
|
|
1221
1294
|
hostWideTools: hostWideTools,
|
|
1222
1295
|
aiActions: window[ACTIONS_GLOBAL] || {},
|
|
1223
1296
|
maxRounds: MAX_ROUNDS,
|
|
1297
|
+
provider: LLM_PROVIDER === "ollama" ? "ollama" : "hub",
|
|
1224
1298
|
signal: currentAbort.signal,
|
|
1299
|
+
onToolCall: function(toolCall) { announceToolCall(toolCall); },
|
|
1300
|
+
onToolDispatched: function(outcome) { resolveToolCall(outcome); },
|
|
1225
1301
|
onPhase: function(name) {
|
|
1226
1302
|
// 'thinking' covers the long silence before the first
|
|
1227
1303
|
// token; 'responding' means text is on its way.
|
|
@@ -1258,23 +1334,15 @@
|
|
|
1258
1334
|
conversation.push({ role: "user", content: userText });
|
|
1259
1335
|
conversation.push({ role: "assistant", content: result.content });
|
|
1260
1336
|
|
|
1261
|
-
//
|
|
1262
|
-
//
|
|
1263
|
-
//
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
var chip = document.createElement("span");
|
|
1271
|
-
chip.className = "lmw-tool-chip" + (d.error ? " error" : "");
|
|
1272
|
-
chip.textContent = (d.error ? "❌ " : "🔧 ") + d.toolCall.name;
|
|
1273
|
-
if (d.error) chip.title = d.error.message;
|
|
1274
|
-
chips.appendChild(chip);
|
|
1275
|
-
});
|
|
1276
|
-
assistantBody.appendChild(chips);
|
|
1277
|
-
}
|
|
1337
|
+
// Anything still pending never reported an outcome (a class that
|
|
1338
|
+
// does not round-trip, or a dispatch that vanished). Reconcile
|
|
1339
|
+
// from the loop's own record so no chip is left spinning.
|
|
1340
|
+
result.dispatched.forEach(function(d) { resolveToolCall(d); });
|
|
1341
|
+
pendingChips.forEach(function(chip) {
|
|
1342
|
+
chip.className = "lmw-tool-chip";
|
|
1343
|
+
chip.textContent = "🔧 " + chip.dataset.tool;
|
|
1344
|
+
});
|
|
1345
|
+
pendingChips = [];
|
|
1278
1346
|
|
|
1279
1347
|
// Skipped = LLM tried a tool that doesn't exist. Real signal.
|
|
1280
1348
|
if (result.skipped.length > 0) {
|