llm_meta_widget 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Rakefile +11 -2
- data/app/assets/javascripts/llm_meta_widget/orchestrator.js +113 -61
- data/app/assets/stylesheets/llm_meta_widget/conversation.css +48 -0
- data/app/helpers/llm_meta_widget/widget_helper.rb +5 -1
- data/app/views/llm_meta_widget/_chat_panel.html.erb +218 -32
- data/lib/llm_meta_widget/version.rb +1 -1
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 20aa5babd648e80015577ece5c815b37f6fe5e06d6136c91647f49831016833b
|
|
4
|
+
data.tar.gz: 8dfc055787110255fbf32368bc3f29cf11b3be178477dac9dadf08243a48d096
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 2545397e740fefb805c8b8b3f9c90ca78814d71b2138f91a35afd38e1499f9bace74428faaf9160cbda4d9a5b51421a81f16f51df82c7ce88389d07acb97293d
|
|
7
|
+
data.tar.gz: a05e51a8b8e94a0153d5c7612901cba9ff2f31beb3142e656852cfa43c87b65ecc513d42c2a6a7a9b2a7ef6b0be3f240ec067bd468afd4b320684089cddfff3b
|
data/Rakefile
CHANGED
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
# `rake build` / `rake release`, used by .github/workflows/gem_release.yml.
|
|
2
2
|
# The widget has no Ruby test suite of its own: its logic is the browser
|
|
3
|
-
# orchestrator, tested with `node --test` (see `rake test_js`)
|
|
3
|
+
# orchestrator, tested with `node --test` (see `rake test_js`) and linted
|
|
4
|
+
# with eslint (`rake lint_js`). The lint covers the panel's script too, by
|
|
5
|
+
# extracting it from the ERB — three bugs reached a browser through that gap
|
|
6
|
+
# before it existed: an undefined variable, a stale import, a redeclared one.
|
|
4
7
|
require "bundler/gem_tasks"
|
|
5
8
|
|
|
6
9
|
desc "Run the orchestrator's JavaScript tests"
|
|
@@ -8,4 +11,10 @@ task :test_js do
|
|
|
8
11
|
sh "node --test app/assets/javascripts/llm_meta_widget/orchestrator.test.mjs"
|
|
9
12
|
end
|
|
10
13
|
|
|
11
|
-
|
|
14
|
+
desc "Lint the widget's JavaScript, including the panel script inside the ERB"
|
|
15
|
+
task :lint_js do
|
|
16
|
+
sh "npm install --silent" unless Dir.exist?("node_modules")
|
|
17
|
+
sh "npm run lint"
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
task default: [ :lint_js, :test_js ]
|
|
@@ -446,8 +446,22 @@ export async function runChatLoop(opts) {
|
|
|
446
446
|
const localOut = await dispatchLocalToolCalls(localCalls, aiActions)
|
|
447
447
|
allDispatched.push(...localOut.dispatched)
|
|
448
448
|
|
|
449
|
-
//
|
|
449
|
+
// A page action's outcome goes back to the model like any other tool
|
|
450
|
+
// result. It used to be dropped, on the grounds that a write to the page
|
|
451
|
+
// has nothing to report — but a turn whose only calls were page actions
|
|
452
|
+
// then ended the loop, so any task that writes to the page and THEN needs
|
|
453
|
+
// a tool ("select these dictionaries, now annotate") was cut off after
|
|
454
|
+
// the write. It also means a failed action is something the model can
|
|
455
|
+
// see and correct, instead of a red mark only the user notices.
|
|
450
456
|
const roundTripResults = []
|
|
457
|
+
for (const { toolCall, error } of localOut.dispatched) {
|
|
458
|
+
roundTripResults.push({
|
|
459
|
+
tc: toolCall,
|
|
460
|
+
result: error ? { error: String(error.message || error) } : { ok: true, applied: toolCall.name }
|
|
461
|
+
})
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
// Class 2: host-wide — direct MCP JSON-RPC POST to the host's own endpoint
|
|
451
465
|
for (const tc of hostWideCalls) {
|
|
452
466
|
const tool = hostWideByName[tc.name]
|
|
453
467
|
const args = coerceArguments(tc.arguments)
|
|
@@ -490,9 +504,10 @@ export async function runChatLoop(opts) {
|
|
|
490
504
|
toolCalls: turnResult.toolCalls
|
|
491
505
|
})
|
|
492
506
|
|
|
493
|
-
// Terminate when there's nothing to feed back
|
|
494
|
-
//
|
|
495
|
-
//
|
|
507
|
+
// Terminate when there's nothing to feed back: a turn with no tool calls
|
|
508
|
+
// at all, or one whose only calls were unknown. Anything that executed —
|
|
509
|
+
// page action, host tool or proxied tool — produces a result the model
|
|
510
|
+
// sees, and gets another round to act on.
|
|
496
511
|
if (roundTripResults.length === 0) {
|
|
497
512
|
return {
|
|
498
513
|
content: turnResult.content,
|
|
@@ -780,38 +795,50 @@ function parseSseFrame(raw) {
|
|
|
780
795
|
// ---- static-primitives extension ---------------------------------------
|
|
781
796
|
//
|
|
782
797
|
// Prototype of `io.modelcontextprotocol/static-primitives` (the SEP-2127
|
|
783
|
-
// follow-on)
|
|
784
|
-
// entry
|
|
785
|
-
//
|
|
786
|
-
//
|
|
798
|
+
// follow-on). Three optional fields a server may declare on a resources/list
|
|
799
|
+
// entry:
|
|
800
|
+
//
|
|
801
|
+
// sizeBytes - byte length of the payload resources/read would return, so a
|
|
802
|
+
// client can decide whether to attach it BEFORE fetching it.
|
|
803
|
+
// volatility - "stable" (content fixed) or "volatile" (varies between
|
|
804
|
+
// turns, so a client re-reads it each turn).
|
|
805
|
+
// autoAttach - may a client attach this without the user asking for it?
|
|
787
806
|
//
|
|
788
|
-
//
|
|
789
|
-
//
|
|
807
|
+
// volatility and autoAttach are separate on purpose. Whether content changes
|
|
808
|
+
// says nothing about whether it may be attached unasked, and the client needs
|
|
809
|
+
// autoAttach as a boolean anyway, because its own byte budget can withhold a
|
|
810
|
+
// resource the server was happy to hand over.
|
|
811
|
+
//
|
|
812
|
+
// All three are optional, and a server declaring none must behave exactly as
|
|
813
|
+
// it did before the extension existed: fetch once, trim to budget, attach on
|
|
814
|
+
// the first turn only.
|
|
790
815
|
export const STATIC_PRIMITIVES_META = "io.modelcontextprotocol/static-primitives"
|
|
791
816
|
|
|
792
|
-
const
|
|
817
|
+
const VOLATILITIES = [ "stable", "volatile" ]
|
|
793
818
|
|
|
794
819
|
export function resourceHints(entry) {
|
|
795
820
|
const meta = (entry && entry._meta && entry._meta[STATIC_PRIMITIVES_META]) || {}
|
|
796
|
-
const hint = meta.attachmentHint
|
|
797
821
|
return {
|
|
798
822
|
// A non-numeric or absent size means "unknown", never 0 — 0 would read
|
|
799
823
|
// as a free resource and sail through every budget check.
|
|
800
824
|
sizeBytes: typeof meta.sizeBytes === "number" && isFinite(meta.sizeBytes) ? meta.sizeBytes : null,
|
|
801
|
-
|
|
825
|
+
volatility: VOLATILITIES.indexOf(meta.volatility) === -1 ? "stable" : meta.volatility,
|
|
826
|
+
autoAttach: typeof meta.autoAttach === "boolean" ? meta.autoAttach : true
|
|
802
827
|
}
|
|
803
828
|
}
|
|
804
829
|
|
|
805
830
|
// The pre-flight decision, made from the resources/list entry alone.
|
|
806
831
|
// `fetch: false` means the bytes never cross the wire at all.
|
|
807
832
|
export function planResourceAttachment(entry, budgetBytes) {
|
|
808
|
-
const { sizeBytes,
|
|
809
|
-
const base = { sizeBytes,
|
|
833
|
+
const { sizeBytes, volatility, autoAttach } = resourceHints(entry)
|
|
834
|
+
const base = { sizeBytes, volatility, uri: entry && entry.uri }
|
|
810
835
|
|
|
811
|
-
if (
|
|
812
|
-
return { ...base, fetch: false, autoAttach: false, reason: "
|
|
836
|
+
if (!autoAttach) {
|
|
837
|
+
return { ...base, fetch: false, autoAttach: false, reason: "not-auto-attach" }
|
|
813
838
|
}
|
|
814
839
|
if (sizeBytes !== null && sizeBytes > budgetBytes) {
|
|
840
|
+
// The client's budget overrides the server's willingness — which is why
|
|
841
|
+
// autoAttach has to be a boolean here rather than a restatement of a hint.
|
|
815
842
|
return { ...base, fetch: false, autoAttach: false, reason: "over-budget" }
|
|
816
843
|
}
|
|
817
844
|
return {
|
|
@@ -824,24 +851,6 @@ export function planResourceAttachment(entry, budgetBytes) {
|
|
|
824
851
|
}
|
|
825
852
|
}
|
|
826
853
|
|
|
827
|
-
// Per-turn gate. Replaces a single `sent` boolean, which silently assumed
|
|
828
|
-
// every resource was "once" and would have re-sent nothing for a resource
|
|
829
|
-
// whose content actually varies between turns.
|
|
830
|
-
export function createResourceAttacher(plan) {
|
|
831
|
-
let attachedOnce = false
|
|
832
|
-
return {
|
|
833
|
-
// Returns whether to attach on THIS turn, and records the answer.
|
|
834
|
-
take() {
|
|
835
|
-
if (!plan || !plan.autoAttach) return false
|
|
836
|
-
if (plan.attachmentHint === "each-turn") return true
|
|
837
|
-
if (attachedOnce) return false
|
|
838
|
-
attachedOnce = true
|
|
839
|
-
return true
|
|
840
|
-
},
|
|
841
|
-
get attachmentHint() { return plan ? plan.attachmentHint : null }
|
|
842
|
-
}
|
|
843
|
-
}
|
|
844
|
-
|
|
845
854
|
// Safety net for a server that declares no size: shrink an oversized
|
|
846
855
|
// catalog to its names rather than dropping it, and hard-truncate anything
|
|
847
856
|
// that is not a recognised catalog shape.
|
|
@@ -858,38 +867,77 @@ export function trimResourceText(text, maxBytes) {
|
|
|
858
867
|
return text.slice(0, maxBytes) + "\n…truncated"
|
|
859
868
|
}
|
|
860
869
|
|
|
861
|
-
//
|
|
862
|
-
//
|
|
863
|
-
//
|
|
864
|
-
|
|
865
|
-
|
|
870
|
+
// Read one resource and shape it for the system prompt. The trim is applied
|
|
871
|
+
// on every read, not just the first: a declared size can go stale, and a
|
|
872
|
+
// volatile resource is a fresh gamble each turn.
|
|
873
|
+
async function readResourceContext({ endpoint, uri, name, read, budgetBytes }) {
|
|
874
|
+
try {
|
|
875
|
+
const result = await read({ endpoint, uri })
|
|
876
|
+
const entry = ((result && result.contents) || [])[0]
|
|
877
|
+
if (!entry || !entry.text) return null
|
|
878
|
+
return { uri, name: name || uri, text: trimResourceText(entry.text, budgetBytes) }
|
|
879
|
+
} catch (e) {
|
|
880
|
+
// Optional context — a failed read must never block the widget.
|
|
881
|
+
return null
|
|
882
|
+
}
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
// The discovery sequence for one endpoint's reference resource, kept here
|
|
886
|
+
// rather than in the panel so the decision AND the call site are testable
|
|
887
|
+
// together: a gate nothing consults is exactly the bug this shape prevents.
|
|
888
|
+
//
|
|
889
|
+
// A volatile resource is NOT read here. Its content is only meaningful for
|
|
890
|
+
// the turn it is attached to, so reading it at boot would buy a copy that is
|
|
891
|
+
// already suspect by the time anyone sends a message.
|
|
866
892
|
export async function loadHostResource({ endpoint, budgetBytes, list, read, onSkip }) {
|
|
867
893
|
const resources = await list({ endpoint })
|
|
868
894
|
const candidate = (resources || []).filter((r) => (r.mimeType || "") === "application/json")[0]
|
|
869
895
|
if (!candidate) return null
|
|
870
896
|
|
|
871
897
|
const plan = planResourceAttachment(candidate, budgetBytes)
|
|
898
|
+
plan.name = candidate.name || candidate.uri
|
|
899
|
+
// Carried so a volatile re-read knows where to go: discovery happens once
|
|
900
|
+
// at boot, but the fetch it authorises happens on every later turn.
|
|
901
|
+
plan.endpoint = endpoint
|
|
872
902
|
if (!plan.fetch) {
|
|
873
903
|
if (onSkip) onSkip(plan)
|
|
874
904
|
return { plan, context: null }
|
|
875
905
|
}
|
|
906
|
+
if (plan.volatility === "volatile") return { plan, context: null }
|
|
876
907
|
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
908
|
+
const context = await readResourceContext({
|
|
909
|
+
endpoint, uri: candidate.uri, name: plan.name, read, budgetBytes
|
|
910
|
+
})
|
|
911
|
+
return { plan, context }
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
export function resourceContextLines(context) {
|
|
915
|
+
if (!context) return []
|
|
916
|
+
return [ "", "Reference data — " + context.name + " (" + context.uri + "):", context.text ]
|
|
917
|
+
}
|
|
918
|
+
|
|
919
|
+
// What to attach on THIS turn.
|
|
920
|
+
//
|
|
921
|
+
// `volatility` governs RE-FETCHING, not re-inclusion: a stable resource is
|
|
922
|
+
// read once and then re-used, but it stays in every turn's system prompt for
|
|
923
|
+
// as long as the conversation lasts. Showing it only on the first turn put
|
|
924
|
+
// the reference data in the one turn that could not use it — the model then
|
|
925
|
+
// spent several tool calls rediscovering what it had already been given,
|
|
926
|
+
// which costs more tokens than simply keeping it.
|
|
927
|
+
//
|
|
928
|
+
// Whether a resource is affordable at all is decided once, from sizeBytes,
|
|
929
|
+
// before it is ever fetched.
|
|
930
|
+
export async function resourceLinesForTurn({ plan, cached, endpoint, read, budgetBytes }) {
|
|
931
|
+
if (!plan || !plan.autoAttach) return { lines: [], cached }
|
|
932
|
+
|
|
933
|
+
if (plan.volatility === "volatile") {
|
|
934
|
+
const fresh = await readResourceContext({
|
|
935
|
+
endpoint, uri: plan.uri, name: plan.name, read, budgetBytes
|
|
936
|
+
})
|
|
937
|
+
return fresh ? { lines: resourceContextLines(fresh), cached: fresh } : { lines: [], cached }
|
|
892
938
|
}
|
|
939
|
+
|
|
940
|
+
return { lines: resourceContextLines(cached), cached }
|
|
893
941
|
}
|
|
894
942
|
|
|
895
943
|
// ---- prompt templates ---------------------------------------------------
|
|
@@ -930,6 +978,16 @@ export function resolvePromptArguments(prompt, state) {
|
|
|
930
978
|
return { args, missing }
|
|
931
979
|
}
|
|
932
980
|
|
|
981
|
+
// What a template will take from the page, and what it will have to ask for.
|
|
982
|
+
// Shown to a first-time visitor so the offer is concrete rather than a bare
|
|
983
|
+
// verb: they can see that the text box is empty before they click.
|
|
984
|
+
export function promptArgumentSummary(prompt, state) {
|
|
985
|
+
return ((prompt && prompt.arguments) || []).map((arg) => {
|
|
986
|
+
const value = promptArgFromState(arg.name, state)
|
|
987
|
+
return { name: arg.name, value, filled: value.length > 0, required: !!arg.required }
|
|
988
|
+
})
|
|
989
|
+
}
|
|
990
|
+
|
|
933
991
|
export function promptButtonProps(prompt) {
|
|
934
992
|
return {
|
|
935
993
|
label: prompt.title || prompt.name,
|
|
@@ -937,9 +995,3 @@ export function promptButtonProps(prompt) {
|
|
|
937
995
|
}
|
|
938
996
|
}
|
|
939
997
|
|
|
940
|
-
// The per-turn attach decision AND its formatting, so the two cannot drift
|
|
941
|
-
// apart: asking whether to attach is what consumes the turn.
|
|
942
|
-
export function nextResourceContextLines(context, attacher) {
|
|
943
|
-
if (!context || !attacher || !attacher.take()) return []
|
|
944
|
-
return [ "", "Reference data — " + context.name + " (" + context.uri + "):", context.text ]
|
|
945
|
-
}
|
|
@@ -181,3 +181,51 @@
|
|
|
181
181
|
max-height: 10em;
|
|
182
182
|
overflow-y: auto;
|
|
183
183
|
}
|
|
184
|
+
|
|
185
|
+
/* ---- "working" signals, ported from llm_meta_chat's chats.css ------------
|
|
186
|
+
*
|
|
187
|
+
* The wait is often long on local models — a tool round re-processes the
|
|
188
|
+
* whole prompt plus the tool results before the first token — so a still
|
|
189
|
+
* label reads as a hang. Same class names as the chat app, so the two
|
|
190
|
+
* surfaces behave identically. */
|
|
191
|
+
.message-role .role-spinner {
|
|
192
|
+
display: inline-block;
|
|
193
|
+
animation: role-working-spin 1.8s linear infinite;
|
|
194
|
+
transform-origin: 50% 50%;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
@keyframes role-working-spin {
|
|
198
|
+
from { transform: rotate(0deg); }
|
|
199
|
+
to { transform: rotate(360deg); }
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/* Staggered three-dot indicator on the reasoning summary while thinking is
|
|
203
|
+
* still streaming. Hidden unless .thinking-active is present. */
|
|
204
|
+
.thinking-dots {
|
|
205
|
+
display: none;
|
|
206
|
+
margin-left: 0.35em;
|
|
207
|
+
letter-spacing: 0.15em;
|
|
208
|
+
font-weight: bold;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
.message-thinking.thinking-active .thinking-dots { display: inline-block; }
|
|
212
|
+
|
|
213
|
+
.message-thinking.thinking-active .thinking-dots span {
|
|
214
|
+
display: inline-block;
|
|
215
|
+
opacity: 0.25;
|
|
216
|
+
animation: thinking-dot 1.4s ease-in-out infinite;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
.message-thinking.thinking-active .thinking-dots span:nth-child(2) { animation-delay: 0.2s; }
|
|
220
|
+
.message-thinking.thinking-active .thinking-dots span:nth-child(3) { animation-delay: 0.4s; }
|
|
221
|
+
|
|
222
|
+
@keyframes thinking-dot {
|
|
223
|
+
0%, 60%, 100% { opacity: 0.25; }
|
|
224
|
+
30% { opacity: 1; }
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/* Respect a reader's motion preference: keep the label, drop the movement. */
|
|
228
|
+
@media (prefers-reduced-motion: reduce) {
|
|
229
|
+
.message-role .role-spinner,
|
|
230
|
+
.message-thinking.thinking-active .thinking-dots span { animation: none; }
|
|
231
|
+
}
|
|
@@ -35,7 +35,11 @@ module LlmMetaWidget
|
|
|
35
35
|
# anon" (all Ollama models / all public_to_anonymous MCP servers).
|
|
36
36
|
# Pass arrays to curate.
|
|
37
37
|
models: nil, # e.g. ["qwen3-6-35b-fast", "qwen3-6-35b-no-think"]
|
|
38
|
-
hub_tools: nil
|
|
38
|
+
hub_tools: nil, # e.g. ["togomcp", "pubdictionaries"] — MCP server names
|
|
39
|
+
# First thing a visitor sees when the panel opens, above the offered
|
|
40
|
+
# prompt templates. nil → a generic line. A blank panel tells a
|
|
41
|
+
# first-time visitor nothing about what the assistant can do for them.
|
|
42
|
+
greeting: nil
|
|
39
43
|
}.freeze
|
|
40
44
|
|
|
41
45
|
def llm_meta_widget(base_url:, model:, **overrides)
|
|
@@ -268,6 +268,59 @@
|
|
|
268
268
|
border-radius: 8px;
|
|
269
269
|
padding: 8px;
|
|
270
270
|
}
|
|
271
|
+
#llm-meta-widget-chat .message-content details > summary {
|
|
272
|
+
cursor: pointer;
|
|
273
|
+
font-weight: 600;
|
|
274
|
+
}
|
|
275
|
+
#llm-meta-widget-chat .lmw-sent-text {
|
|
276
|
+
margin-top: 6px;
|
|
277
|
+
font-size: 12px;
|
|
278
|
+
opacity: 0.8;
|
|
279
|
+
white-space: pre-wrap;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/* Opening state: greeting + the offers themselves. Shown until the first
|
|
283
|
+
* real turn, and again after Clear. */
|
|
284
|
+
#llm-meta-widget-chat .lmw-welcome { padding: 4px 2px 2px; }
|
|
285
|
+
#llm-meta-widget-chat .lmw-welcome-hello {
|
|
286
|
+
margin: 0 0 12px;
|
|
287
|
+
font-size: 14px;
|
|
288
|
+
line-height: 1.5;
|
|
289
|
+
color: #374151;
|
|
290
|
+
}
|
|
291
|
+
#llm-meta-widget-chat .lmw-welcome-card {
|
|
292
|
+
background-color: #f9fafb;
|
|
293
|
+
border: 1px solid #e5e7eb;
|
|
294
|
+
border-radius: 8px;
|
|
295
|
+
padding: 10px 12px;
|
|
296
|
+
margin-bottom: 8px;
|
|
297
|
+
}
|
|
298
|
+
#llm-meta-widget-chat .lmw-welcome-start {
|
|
299
|
+
background-color: #eff6ff;
|
|
300
|
+
color: #1d4ed8;
|
|
301
|
+
border: 1px solid #bfdbfe;
|
|
302
|
+
border-radius: 999px;
|
|
303
|
+
padding: 4px 12px;
|
|
304
|
+
font-size: 13px;
|
|
305
|
+
font-family: inherit;
|
|
306
|
+
cursor: pointer;
|
|
307
|
+
}
|
|
308
|
+
#llm-meta-widget-chat .lmw-welcome-start:hover { background-color: #dbeafe; }
|
|
309
|
+
#llm-meta-widget-chat .lmw-welcome-why {
|
|
310
|
+
margin: 8px 0 0;
|
|
311
|
+
font-size: 13px;
|
|
312
|
+
line-height: 1.45;
|
|
313
|
+
color: #4b5563;
|
|
314
|
+
}
|
|
315
|
+
#llm-meta-widget-chat .lmw-welcome-slots {
|
|
316
|
+
margin: 8px 0 0;
|
|
317
|
+
padding-left: 16px;
|
|
318
|
+
font-size: 12px;
|
|
319
|
+
color: #6b7280;
|
|
320
|
+
}
|
|
321
|
+
#llm-meta-widget-chat .lmw-welcome-slots li { margin-bottom: 2px; }
|
|
322
|
+
#llm-meta-widget-chat .lmw-welcome-slots li.filled { color: #047857; }
|
|
323
|
+
|
|
271
324
|
/* Server-offered prompt templates. The row is display:none in markup and
|
|
272
325
|
* gets its display back only when a prompt actually exists, so the inline
|
|
273
326
|
* style and this rule have to agree on "flex". */
|
|
@@ -398,8 +451,8 @@
|
|
|
398
451
|
<script type="module">
|
|
399
452
|
import { runChatLoop, fetchMcpManifest, listMcpPrompts, getMcpPrompt,
|
|
400
453
|
listMcpResources, readMcpResource, promptMessagesToText,
|
|
401
|
-
loadHostResource,
|
|
402
|
-
promptButtonProps,
|
|
454
|
+
loadHostResource, resourceLinesForTurn, resolvePromptArguments,
|
|
455
|
+
promptButtonProps, promptArgumentSummary } from "<%= orchestrator_path %>";
|
|
403
456
|
import { marked } from "/llm_meta_widget_assets/marked.esm.js";
|
|
404
457
|
|
|
405
458
|
// Standard prose settings — GFM (tables, autolinks, strikethrough),
|
|
@@ -411,6 +464,7 @@
|
|
|
411
464
|
var API_KEY_UUID = <%= api_key_uuid.to_json.html_safe %>;
|
|
412
465
|
var MODEL = <%= model.to_json.html_safe %>;
|
|
413
466
|
var ACTIONS_SCHEMA_ID = <%= actions_schema_id.to_json.html_safe %>;
|
|
467
|
+
var GREETING = <%= greeting.to_json.html_safe %>;
|
|
414
468
|
var STATE_GLOBAL = <%= state_global.to_json.html_safe %>;
|
|
415
469
|
var ACTIONS_GLOBAL = <%= actions_global.to_json.html_safe %>;
|
|
416
470
|
var REMOTE_TOOLS_SCHEMA_ID = <%= remote_tools_schema_id.to_json.html_safe %>;
|
|
@@ -755,7 +809,6 @@
|
|
|
755
809
|
var hostWideTools = [];
|
|
756
810
|
var hostWidePrompts = [];
|
|
757
811
|
var resourceContext = null; // payload of the host's reference resource
|
|
758
|
-
var resourceAttacher = null; // per-turn gate, built from the server's hint
|
|
759
812
|
var resourcePlan = null; // the pre-flight decision, kept so Clear can re-arm the gate
|
|
760
813
|
|
|
761
814
|
// Beyond ~2k tokens a reference resource starts crowding out the
|
|
@@ -763,9 +816,6 @@
|
|
|
763
816
|
// is 218 dictionaries / ~35KB / ~10k tokens.
|
|
764
817
|
var RESOURCE_BUDGET_BYTES = 8000;
|
|
765
818
|
|
|
766
|
-
// Set when a server declared a resource we deliberately did not attach,
|
|
767
|
-
// so the reason is inspectable rather than a silent absence.
|
|
768
|
-
var resourceSkip = null;
|
|
769
819
|
|
|
770
820
|
var wellKnownReady = (async function() {
|
|
771
821
|
var urls = WELL_KNOWN_URLS === null
|
|
@@ -789,7 +839,7 @@
|
|
|
789
839
|
var prompts = await listMcpPrompts({ endpoint: endpoint });
|
|
790
840
|
hostWidePrompts = hostWidePrompts.concat(prompts);
|
|
791
841
|
|
|
792
|
-
if (
|
|
842
|
+
if (resourcePlan === null) {
|
|
793
843
|
// Listing, size gate and read all live in the orchestrator, where
|
|
794
844
|
// they are tested together — a gate nothing consults is exactly
|
|
795
845
|
// the bug this shape prevents.
|
|
@@ -799,19 +849,21 @@
|
|
|
799
849
|
list: listMcpResources,
|
|
800
850
|
read: readMcpResource,
|
|
801
851
|
onSkip: function(plan) {
|
|
802
|
-
resourceSkip = plan;
|
|
803
852
|
console.info("[llm_meta_widget] not attaching " + plan.uri +
|
|
804
853
|
" (" + plan.reason + ", sizeBytes=" + plan.sizeBytes + ")");
|
|
805
854
|
}
|
|
806
855
|
});
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
856
|
+
// A volatile resource comes back with no context — it is read
|
|
857
|
+
// per turn instead — so the plan, not the payload, is what says
|
|
858
|
+
// whether this endpoint offered anything worth attaching.
|
|
859
|
+
if (loaded && loaded.plan && loaded.plan.fetch) {
|
|
860
|
+
resourceContext = loaded.context;
|
|
861
|
+
resourcePlan = loaded.plan;
|
|
811
862
|
}
|
|
812
863
|
}
|
|
813
864
|
}
|
|
814
865
|
renderPromptButtons();
|
|
866
|
+
renderWelcome();
|
|
815
867
|
})();
|
|
816
868
|
|
|
817
869
|
// ---- server-offered prompt templates --------------------------------
|
|
@@ -841,6 +893,66 @@
|
|
|
841
893
|
promptsEl.style.display = "";
|
|
842
894
|
}
|
|
843
895
|
|
|
896
|
+
// A blank panel tells a first-time visitor nothing. Open with a greeting
|
|
897
|
+
// and the offers themselves — each template showing what it will take from
|
|
898
|
+
// the page and what it will ask for — so the assistant is the page's way
|
|
899
|
+
// in rather than a box you must already know how to talk to.
|
|
900
|
+
function renderWelcome() {
|
|
901
|
+
if (!historyEl || historyEl.querySelector(".message")) return;
|
|
902
|
+
historyEl.textContent = "";
|
|
903
|
+
|
|
904
|
+
var box = document.createElement("div");
|
|
905
|
+
box.className = "lmw-welcome";
|
|
906
|
+
|
|
907
|
+
var hello = document.createElement("p");
|
|
908
|
+
hello.className = "lmw-welcome-hello";
|
|
909
|
+
hello.textContent = GREETING ||
|
|
910
|
+
"Hi — I can work this page for you. Tell me what you need in your own words" +
|
|
911
|
+
(hostWidePrompts.length ? ", or start with one of these:" : ".");
|
|
912
|
+
box.appendChild(hello);
|
|
913
|
+
|
|
914
|
+
var state = window[STATE_GLOBAL] || {};
|
|
915
|
+
hostWidePrompts.forEach(function(prompt) {
|
|
916
|
+
var props = promptButtonProps(prompt);
|
|
917
|
+
var card = document.createElement("div");
|
|
918
|
+
card.className = "lmw-welcome-card";
|
|
919
|
+
|
|
920
|
+
var start = document.createElement("button");
|
|
921
|
+
start.type = "button";
|
|
922
|
+
start.className = "lmw-welcome-start";
|
|
923
|
+
start.textContent = props.label;
|
|
924
|
+
start.addEventListener("click", function() { runPromptTemplate(prompt, start); });
|
|
925
|
+
card.appendChild(start);
|
|
926
|
+
|
|
927
|
+
if (prompt.description) {
|
|
928
|
+
var why = document.createElement("p");
|
|
929
|
+
why.className = "lmw-welcome-why";
|
|
930
|
+
why.textContent = prompt.description;
|
|
931
|
+
card.appendChild(why);
|
|
932
|
+
}
|
|
933
|
+
|
|
934
|
+
// Name the placeholders, filled or not, so the offer is concrete:
|
|
935
|
+
// a newcomer can see the text box is empty before clicking.
|
|
936
|
+
var summary = promptArgumentSummary(prompt, state);
|
|
937
|
+
if (summary.length) {
|
|
938
|
+
var slots = document.createElement("ul");
|
|
939
|
+
slots.className = "lmw-welcome-slots";
|
|
940
|
+
summary.forEach(function(slot) {
|
|
941
|
+
var li = document.createElement("li");
|
|
942
|
+
li.className = slot.filled ? "filled" : "empty";
|
|
943
|
+
li.textContent = slot.filled
|
|
944
|
+
? slot.name + ": " + (slot.value.length > 60 ? slot.value.slice(0, 60) + "…" : slot.value)
|
|
945
|
+
: slot.name + ": not set yet — I'll ask, or work it out";
|
|
946
|
+
slots.appendChild(li);
|
|
947
|
+
});
|
|
948
|
+
card.appendChild(slots);
|
|
949
|
+
}
|
|
950
|
+
box.appendChild(card);
|
|
951
|
+
});
|
|
952
|
+
|
|
953
|
+
historyEl.appendChild(box);
|
|
954
|
+
}
|
|
955
|
+
|
|
844
956
|
async function runPromptTemplate(prompt, button) {
|
|
845
957
|
var resolved = resolvePromptArguments(prompt, window[STATE_GLOBAL] || {});
|
|
846
958
|
var args = resolved.args;
|
|
@@ -860,6 +972,7 @@
|
|
|
860
972
|
var text = promptMessagesToText(result);
|
|
861
973
|
if (!text) { appendTurn("error", "The server returned an empty prompt."); return; }
|
|
862
974
|
inputEl.value = text;
|
|
975
|
+
pendingTurnLabel = promptButtonProps(prompt).label;
|
|
863
976
|
if (typeof formEl.requestSubmit === "function") formEl.requestSubmit();
|
|
864
977
|
else formEl.dispatchEvent(new Event("submit", { cancelable: true }));
|
|
865
978
|
} catch (e) {
|
|
@@ -869,13 +982,22 @@
|
|
|
869
982
|
}
|
|
870
983
|
}
|
|
871
984
|
|
|
872
|
-
|
|
985
|
+
// Set just before a template submits, so its turn is labelled by what the
|
|
986
|
+
// user actually did — "Annotate text" — instead of showing a paragraph of
|
|
987
|
+
// server-written instructions as though they had typed it. The instructions
|
|
988
|
+
// stay one click away rather than hidden: what was sent is what is shown.
|
|
989
|
+
var pendingTurnLabel = null;
|
|
990
|
+
|
|
991
|
+
function appendTurn(role, text, turnLabel) {
|
|
873
992
|
// Class names mirror llm_meta_chat's chats/_message.html.erb —
|
|
874
993
|
// `.message.<role>`, `.message-role`, `.message-content` — so the
|
|
875
994
|
// shared conversation.css styles apply directly (see the <link>
|
|
876
995
|
// above). The .lmw-* prefix is reserved for widget-CHROME classes
|
|
877
996
|
// (header, clear button, scroll region, input area) that aren't
|
|
878
997
|
// part of the shared conversation surface.
|
|
998
|
+
var welcome = historyEl.querySelector(".lmw-welcome");
|
|
999
|
+
if (welcome) welcome.remove();
|
|
1000
|
+
|
|
879
1001
|
var div = document.createElement("div");
|
|
880
1002
|
div.className = "message " + role;
|
|
881
1003
|
var label = document.createElement("div");
|
|
@@ -888,12 +1010,25 @@
|
|
|
888
1010
|
// content is rendered as markdown but only after the assistant's
|
|
889
1011
|
// text is streamed in via renderMarkdownInto — this appendTurn
|
|
890
1012
|
// creates the empty container.
|
|
891
|
-
|
|
1013
|
+
if (turnLabel) {
|
|
1014
|
+
var details = document.createElement("details");
|
|
1015
|
+
var summary = document.createElement("summary");
|
|
1016
|
+
summary.textContent = turnLabel;
|
|
1017
|
+
var full = document.createElement("div");
|
|
1018
|
+
full.className = "lmw-sent-text";
|
|
1019
|
+
full.textContent = text;
|
|
1020
|
+
details.appendChild(summary);
|
|
1021
|
+
details.appendChild(full);
|
|
1022
|
+
body.appendChild(details);
|
|
1023
|
+
} else {
|
|
1024
|
+
body.textContent = text;
|
|
1025
|
+
}
|
|
892
1026
|
div.appendChild(label);
|
|
893
1027
|
div.appendChild(document.createTextNode(" "));
|
|
894
1028
|
div.appendChild(body);
|
|
895
1029
|
historyEl.appendChild(div);
|
|
896
1030
|
historyEl.scrollTop = historyEl.scrollHeight;
|
|
1031
|
+
body.roleLabel = label; // so a turn in flight can show it is working
|
|
897
1032
|
return body;
|
|
898
1033
|
}
|
|
899
1034
|
|
|
@@ -918,6 +1053,23 @@
|
|
|
918
1053
|
return role;
|
|
919
1054
|
}
|
|
920
1055
|
|
|
1056
|
+
// While a turn is in flight the assistant's label becomes a turning gear.
|
|
1057
|
+
// Same markup and class names as llm_meta_chat's message_stream_controller,
|
|
1058
|
+
// so the shared conversation.css drives both. Without it there is no sign
|
|
1059
|
+
// whether the assistant is still working or has quietly stopped — and on a
|
|
1060
|
+
// local model a tool round can take a long time before the first token.
|
|
1061
|
+
function markWorking(label) {
|
|
1062
|
+
if (!label || label.classList.contains("is-working")) return;
|
|
1063
|
+
label.innerHTML = '<span class="role-spinner" aria-hidden="true">\u2699\uFE0F</span> Working…';
|
|
1064
|
+
label.classList.add("is-working");
|
|
1065
|
+
}
|
|
1066
|
+
|
|
1067
|
+
function markDone(label) {
|
|
1068
|
+
if (!label || !label.classList.contains("is-working")) return;
|
|
1069
|
+
label.classList.remove("is-working");
|
|
1070
|
+
label.textContent = roleLabel("assistant");
|
|
1071
|
+
}
|
|
1072
|
+
|
|
921
1073
|
var currentThinkingBlock = null;
|
|
922
1074
|
var currentThinkingBody = null;
|
|
923
1075
|
|
|
@@ -928,8 +1080,17 @@
|
|
|
928
1080
|
var details = document.createElement("details");
|
|
929
1081
|
details.className = "message-thinking";
|
|
930
1082
|
details.open = true;
|
|
1083
|
+
details.classList.add("thinking-active");
|
|
931
1084
|
var summary = document.createElement("summary");
|
|
932
1085
|
summary.textContent = "🤔 thinking…";
|
|
1086
|
+
var dots = document.createElement("span");
|
|
1087
|
+
dots.className = "thinking-dots";
|
|
1088
|
+
for (var i = 0; i < 3; i++) {
|
|
1089
|
+
var dot = document.createElement("span");
|
|
1090
|
+
dot.textContent = ".";
|
|
1091
|
+
dots.appendChild(dot);
|
|
1092
|
+
}
|
|
1093
|
+
summary.appendChild(dots);
|
|
933
1094
|
var body = document.createElement("div");
|
|
934
1095
|
body.className = "message-thinking-content";
|
|
935
1096
|
details.appendChild(summary);
|
|
@@ -943,6 +1104,7 @@
|
|
|
943
1104
|
|
|
944
1105
|
function collapseThinkingBlock() {
|
|
945
1106
|
if (currentThinkingBlock) {
|
|
1107
|
+
currentThinkingBlock.classList.remove("thinking-active");
|
|
946
1108
|
currentThinkingBlock.open = false;
|
|
947
1109
|
var summary = currentThinkingBlock.querySelector("summary");
|
|
948
1110
|
if (summary) summary.textContent = "🤔 thinking (finished)";
|
|
@@ -960,28 +1122,35 @@
|
|
|
960
1122
|
return out;
|
|
961
1123
|
}
|
|
962
1124
|
|
|
963
|
-
function currentSystemPrompt() {
|
|
1125
|
+
function currentSystemPrompt(resourceLines) {
|
|
964
1126
|
return [
|
|
965
1127
|
"You are integrated into a web page as an AI assistant. You have tools available to change page state or fetch information.",
|
|
966
1128
|
"",
|
|
967
1129
|
"RULES for tool use:",
|
|
968
1130
|
"1. If the user's question can be answered from the Current page state below, answer directly with a plain-text response — do NOT invoke a tool.",
|
|
969
1131
|
"2. If the user requests a state change, or needs information not in the page state, invoke the matching tool via a function call. Do not describe your intent in text without actually invoking (a textual promise like \"I will add X\" is a failure).",
|
|
970
|
-
"3. After a tool returns a result,
|
|
971
|
-
"4. NEVER
|
|
1132
|
+
"3. After a tool returns a result, use it: either take the next step the task needs, or — if the task is done — answer in plain text. Do not stop silently after a tool call.",
|
|
1133
|
+
"4. NEVER repeat a call you have already made with the same or similar arguments — its result is already in the conversation history.",
|
|
972
1134
|
"",
|
|
973
1135
|
"Current page state:",
|
|
974
1136
|
JSON.stringify(currentPageState(), null, 2)
|
|
975
|
-
].concat(
|
|
1137
|
+
].concat(resourceLines || []).join("\n");
|
|
976
1138
|
}
|
|
977
1139
|
|
|
978
|
-
//
|
|
979
|
-
//
|
|
980
|
-
//
|
|
981
|
-
//
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
1140
|
+
// Runs once per send, before the system prompt is built: a volatile
|
|
1141
|
+
// resource is re-read here, a stable one re-uses the copy taken at boot.
|
|
1142
|
+
// Either way it is attached to every turn — the server's hint governs
|
|
1143
|
+
// re-FETCHING; how often to include it is this client's call.
|
|
1144
|
+
async function resourceLinesForThisTurn() {
|
|
1145
|
+
var turn = await resourceLinesForTurn({
|
|
1146
|
+
plan: resourcePlan,
|
|
1147
|
+
cached: resourceContext,
|
|
1148
|
+
endpoint: resourcePlan && resourcePlan.endpoint,
|
|
1149
|
+
read: readMcpResource,
|
|
1150
|
+
budgetBytes: RESOURCE_BUDGET_BYTES
|
|
1151
|
+
});
|
|
1152
|
+
resourceContext = turn.cached;
|
|
1153
|
+
return turn.lines;
|
|
985
1154
|
}
|
|
986
1155
|
|
|
987
1156
|
// Per-turn AbortController — lets the Clear button (or a new submit)
|
|
@@ -992,9 +1161,7 @@
|
|
|
992
1161
|
if (currentAbort) { try { currentAbort.abort(); } catch (e) { /* noop */ } }
|
|
993
1162
|
conversation = [];
|
|
994
1163
|
historyEl.innerHTML = "";
|
|
995
|
-
|
|
996
|
-
// again — the old boolean stayed latched and silently withheld it.
|
|
997
|
-
if (resourcePlan) resourceAttacher = createResourceAttacher(resourcePlan);
|
|
1164
|
+
renderWelcome();
|
|
998
1165
|
});
|
|
999
1166
|
|
|
1000
1167
|
// Enter submits, Shift+Enter inserts a newline — matches llm_meta_chat's
|
|
@@ -1008,17 +1175,19 @@
|
|
|
1008
1175
|
}
|
|
1009
1176
|
});
|
|
1010
1177
|
|
|
1011
|
-
formEl.addEventListener("submit", async function(
|
|
1012
|
-
|
|
1178
|
+
formEl.addEventListener("submit", async function(event) {
|
|
1179
|
+
event.preventDefault();
|
|
1013
1180
|
var userText = inputEl.value.trim();
|
|
1014
1181
|
if (!userText) return;
|
|
1015
1182
|
inputEl.value = "";
|
|
1016
|
-
appendTurn("user", userText);
|
|
1183
|
+
appendTurn("user", userText, pendingTurnLabel);
|
|
1184
|
+
pendingTurnLabel = null;
|
|
1017
1185
|
|
|
1018
1186
|
if (currentAbort) { try { currentAbort.abort(); } catch (e) { /* noop */ } }
|
|
1019
1187
|
currentAbort = new AbortController();
|
|
1020
1188
|
|
|
1021
1189
|
var assistantBody = appendTurn("assistant", "");
|
|
1190
|
+
markWorking(assistantBody.roleLabel);
|
|
1022
1191
|
var assistantMarkdown = ""; // accumulate raw markdown, re-render on each delta
|
|
1023
1192
|
|
|
1024
1193
|
try {
|
|
@@ -1028,7 +1197,8 @@
|
|
|
1028
1197
|
// it as sent.
|
|
1029
1198
|
await wellKnownReady;
|
|
1030
1199
|
|
|
1031
|
-
var
|
|
1200
|
+
var resourceLines = await resourceLinesForThisTurn();
|
|
1201
|
+
var messages = [{ role: "system", content: currentSystemPrompt(resourceLines) }]
|
|
1032
1202
|
.concat(conversation)
|
|
1033
1203
|
.concat([{ role: "user", content: userText }]);
|
|
1034
1204
|
|
|
@@ -1043,7 +1213,16 @@
|
|
|
1043
1213
|
aiActions: window[ACTIONS_GLOBAL] || {},
|
|
1044
1214
|
maxRounds: MAX_ROUNDS,
|
|
1045
1215
|
signal: currentAbort.signal,
|
|
1216
|
+
onPhase: function(name) {
|
|
1217
|
+
// 'thinking' covers the long silence before the first
|
|
1218
|
+
// token; 'responding' means text is on its way.
|
|
1219
|
+
if (name === "responding") markDone(assistantBody.roleLabel);
|
|
1220
|
+
else markWorking(assistantBody.roleLabel);
|
|
1221
|
+
},
|
|
1046
1222
|
onRoundStart: function(roundIdx) {
|
|
1223
|
+
// A new round means more work: tool results are going back
|
|
1224
|
+
// to the model, which is the longest wait of all.
|
|
1225
|
+
markWorking(assistantBody.roleLabel);
|
|
1047
1226
|
// Loop mechanics are debugging info, not user-facing signal.
|
|
1048
1227
|
// Reuse the same assistant bubble across rounds — text just
|
|
1049
1228
|
// keeps streaming into it (accumulating markdown). Weaker
|
|
@@ -1060,6 +1239,7 @@
|
|
|
1060
1239
|
historyEl.scrollTop = historyEl.scrollHeight;
|
|
1061
1240
|
},
|
|
1062
1241
|
onTextDelta: function(delta) {
|
|
1242
|
+
markDone(assistantBody.roleLabel);
|
|
1063
1243
|
collapseThinkingBlock();
|
|
1064
1244
|
assistantMarkdown += delta;
|
|
1065
1245
|
renderMarkdownInto(assistantBody, assistantMarkdown);
|
|
@@ -1110,6 +1290,12 @@
|
|
|
1110
1290
|
appendTurn("error", err.message);
|
|
1111
1291
|
}
|
|
1112
1292
|
} finally {
|
|
1293
|
+
// Whatever happened — answered, aborted, threw, or returned
|
|
1294
|
+
// nothing at all — the turn is over and the label must stop
|
|
1295
|
+
// claiming otherwise. A spinner left running is a worse lie than
|
|
1296
|
+
// no spinner: it says "still working" about a turn that ended.
|
|
1297
|
+
markDone(assistantBody.roleLabel);
|
|
1298
|
+
collapseThinkingBlock();
|
|
1113
1299
|
currentAbort = null;
|
|
1114
1300
|
}
|
|
1115
1301
|
});
|