llm_meta_widget 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: f32625e105ab751f473e79517d113c8b84465e676cddaab0f3db861c8f3e563e
4
- data.tar.gz: 249d09f198b93e1b8382b2f5ef719736b260fc4c0c2f53d77e68c203b8e236b2
3
+ metadata.gz: 20aa5babd648e80015577ece5c815b37f6fe5e06d6136c91647f49831016833b
4
+ data.tar.gz: 8dfc055787110255fbf32368bc3f29cf11b3be178477dac9dadf08243a48d096
5
5
  SHA512:
6
- metadata.gz: 841f81296bed31dbe0ebd7c67253b08990be954a0e5b6939f0563c3d97cb525b8826654149350dbae4afd74ce5b61750058f289c9df25ed4fca5b577783db72b
7
- data.tar.gz: ced44aaf75dc424014f878360a4f769247cda1856abb7ffde7e3d66e829e8ae953e1c5b0ba21aea6521abab19b10ac821631e490d9ca47cba7876ea41b5ef8d1
6
+ metadata.gz: 2545397e740fefb805c8b8b3f9c90ca78814d71b2138f91a35afd38e1499f9bace74428faaf9160cbda4d9a5b51421a81f16f51df82c7ce88389d07acb97293d
7
+ data.tar.gz: a05e51a8b8e94a0153d5c7612901cba9ff2f31beb3142e656852cfa43c87b65ecc513d42c2a6a7a9b2a7ef6b0be3f240ec067bd468afd4b320684089cddfff3b
data/Rakefile CHANGED
@@ -1,6 +1,9 @@
1
1
  # `rake build` / `rake release`, used by .github/workflows/gem_release.yml.
2
2
  # The widget has no Ruby test suite of its own: its logic is the browser
3
- # orchestrator, tested with `node --test` (see `rake test_js`).
3
+ # orchestrator, tested with `node --test` (see `rake test_js`) and linted
4
+ # with eslint (`rake lint_js`). The lint covers the panel's script too, by
5
+ # extracting it from the ERB — three bugs reached a browser through that gap
6
+ # before it existed: an undefined variable, a stale import, a redeclared one.
4
7
  require "bundler/gem_tasks"
5
8
 
6
9
  desc "Run the orchestrator's JavaScript tests"
@@ -8,4 +11,10 @@ task :test_js do
8
11
  sh "node --test app/assets/javascripts/llm_meta_widget/orchestrator.test.mjs"
9
12
  end
10
13
 
11
- task default: :test_js
14
+ desc "Lint the widget's JavaScript, including the panel script inside the ERB"
15
+ task :lint_js do
16
+ sh "npm install --silent" unless Dir.exist?("node_modules")
17
+ sh "npm run lint"
18
+ end
19
+
20
+ task default: [ :lint_js, :test_js ]
@@ -446,8 +446,22 @@ export async function runChatLoop(opts) {
446
446
  const localOut = await dispatchLocalToolCalls(localCalls, aiActions)
447
447
  allDispatched.push(...localOut.dispatched)
448
448
 
449
- // Class 2: host-wide — direct MCP JSON-RPC POST to the host's own endpoint
449
+ // A page action's outcome goes back to the model like any other tool
450
+ // result. It used to be dropped, on the grounds that a write to the page
451
+ // has nothing to report — but a turn whose only calls were page actions
452
+ // then ended the loop, so any task that writes to the page and THEN needs
453
+ // a tool ("select these dictionaries, now annotate") was cut off after
454
+ // the write. It also means a failed action is something the model can
455
+ // see and correct, instead of a red mark only the user notices.
450
456
  const roundTripResults = []
457
+ for (const { toolCall, error } of localOut.dispatched) {
458
+ roundTripResults.push({
459
+ tc: toolCall,
460
+ result: error ? { error: String(error.message || error) } : { ok: true, applied: toolCall.name }
461
+ })
462
+ }
463
+
464
+ // Class 2: host-wide — direct MCP JSON-RPC POST to the host's own endpoint
451
465
  for (const tc of hostWideCalls) {
452
466
  const tool = hostWideByName[tc.name]
453
467
  const args = coerceArguments(tc.arguments)
@@ -490,9 +504,10 @@ export async function runChatLoop(opts) {
490
504
  toolCalls: turnResult.toolCalls
491
505
  })
492
506
 
493
- // Terminate when there's nothing to feed back. Locals (Class 3) are
494
- // fire-and-forget; unknowns can't be handled; only Class 2 + Class 1
495
- // execution produces tool results the LLM should see.
507
+ // Terminate when there's nothing to feed back: a turn with no tool calls
508
+ // at all, or one whose only calls were unknown. Anything that executed —
509
+ // page action, host tool or proxied tool — produces a result the model
510
+ // sees, and gets another round to act on.
496
511
  if (roundTripResults.length === 0) {
497
512
  return {
498
513
  content: turnResult.content,
@@ -780,38 +795,50 @@ function parseSseFrame(raw) {
780
795
  // ---- static-primitives extension ---------------------------------------
781
796
  //
782
797
  // Prototype of `io.modelcontextprotocol/static-primitives` (the SEP-2127
783
- // follow-on): two optional fields a server may declare on a resources/list
784
- // entry. `sizeBytes` is the byte length of the payload resources/read would
785
- // return, so a client can decide whether to attach it BEFORE fetching it.
786
- // `attachmentHint` says how often attaching is worth it.
798
+ // follow-on). Three optional fields a server may declare on a resources/list
799
+ // entry:
800
+ //
801
+ // sizeBytes - byte length of the payload resources/read would return, so a
802
+ // client can decide whether to attach it BEFORE fetching it.
803
+ // volatility - "stable" (content fixed) or "volatile" (varies between
804
+ // turns, so a client re-reads it each turn).
805
+ // autoAttach - may a client attach this without the user asking for it?
787
806
  //
788
- // Both are optional, and a server that declares neither must behave exactly
789
- // as it did before the extension existed: fetch, trim to budget, attach once.
807
+ // volatility and autoAttach are separate on purpose. Whether content changes
808
+ // says nothing about whether it may be attached unasked, and the client needs
809
+ // autoAttach as a boolean anyway, because its own byte budget can withhold a
810
+ // resource the server was happy to hand over.
811
+ //
812
+ // All three are optional, and a server declaring none must behave exactly as
813
+ // it did before the extension existed: fetch once, trim to budget, attach on
814
+ // the first turn only.
790
815
  export const STATIC_PRIMITIVES_META = "io.modelcontextprotocol/static-primitives"
791
816
 
792
- const ATTACHMENT_HINTS = [ "once", "each-turn", "on-demand" ]
817
+ const VOLATILITIES = [ "stable", "volatile" ]
793
818
 
794
819
  export function resourceHints(entry) {
795
820
  const meta = (entry && entry._meta && entry._meta[STATIC_PRIMITIVES_META]) || {}
796
- const hint = meta.attachmentHint
797
821
  return {
798
822
  // A non-numeric or absent size means "unknown", never 0 — 0 would read
799
823
  // as a free resource and sail through every budget check.
800
824
  sizeBytes: typeof meta.sizeBytes === "number" && isFinite(meta.sizeBytes) ? meta.sizeBytes : null,
801
- attachmentHint: ATTACHMENT_HINTS.indexOf(hint) === -1 ? "once" : hint
825
+ volatility: VOLATILITIES.indexOf(meta.volatility) === -1 ? "stable" : meta.volatility,
826
+ autoAttach: typeof meta.autoAttach === "boolean" ? meta.autoAttach : true
802
827
  }
803
828
  }
804
829
 
805
830
  // The pre-flight decision, made from the resources/list entry alone.
806
831
  // `fetch: false` means the bytes never cross the wire at all.
807
832
  export function planResourceAttachment(entry, budgetBytes) {
808
- const { sizeBytes, attachmentHint } = resourceHints(entry)
809
- const base = { sizeBytes, attachmentHint, uri: entry && entry.uri }
833
+ const { sizeBytes, volatility, autoAttach } = resourceHints(entry)
834
+ const base = { sizeBytes, volatility, uri: entry && entry.uri }
810
835
 
811
- if (attachmentHint === "on-demand") {
812
- return { ...base, fetch: false, autoAttach: false, reason: "on-demand" }
836
+ if (!autoAttach) {
837
+ return { ...base, fetch: false, autoAttach: false, reason: "not-auto-attach" }
813
838
  }
814
839
  if (sizeBytes !== null && sizeBytes > budgetBytes) {
840
+ // The client's budget overrides the server's willingness — which is why
841
+ // autoAttach has to be a boolean here rather than a restatement of a hint.
815
842
  return { ...base, fetch: false, autoAttach: false, reason: "over-budget" }
816
843
  }
817
844
  return {
@@ -824,24 +851,6 @@ export function planResourceAttachment(entry, budgetBytes) {
824
851
  }
825
852
  }
826
853
 
827
- // Per-turn gate. Replaces a single `sent` boolean, which silently assumed
828
- // every resource was "once" and would have re-sent nothing for a resource
829
- // whose content actually varies between turns.
830
- export function createResourceAttacher(plan) {
831
- let attachedOnce = false
832
- return {
833
- // Returns whether to attach on THIS turn, and records the answer.
834
- take() {
835
- if (!plan || !plan.autoAttach) return false
836
- if (plan.attachmentHint === "each-turn") return true
837
- if (attachedOnce) return false
838
- attachedOnce = true
839
- return true
840
- },
841
- get attachmentHint() { return plan ? plan.attachmentHint : null }
842
- }
843
- }
844
-
845
854
  // Safety net for a server that declares no size: shrink an oversized
846
855
  // catalog to its names rather than dropping it, and hard-truncate anything
847
856
  // that is not a recognised catalog shape.
@@ -858,38 +867,77 @@ export function trimResourceText(text, maxBytes) {
858
867
  return text.slice(0, maxBytes) + "\n…truncated"
859
868
  }
860
869
 
861
- // The whole discovery sequence for one endpoint's reference resource, kept
862
- // here rather than in the panel so the decision AND the call site are
863
- // testable together: a gate that nothing consults is the failure mode this
864
- // replaces. `list` and `read` are injected so a test can assert that an
865
- // over-budget resource is never read.
870
+ // Read one resource and shape it for the system prompt. The trim is applied
871
+ // on every read, not just the first: a declared size can go stale, and a
872
+ // volatile resource is a fresh gamble each turn.
873
+ async function readResourceContext({ endpoint, uri, name, read, budgetBytes }) {
874
+ try {
875
+ const result = await read({ endpoint, uri })
876
+ const entry = ((result && result.contents) || [])[0]
877
+ if (!entry || !entry.text) return null
878
+ return { uri, name: name || uri, text: trimResourceText(entry.text, budgetBytes) }
879
+ } catch (e) {
880
+ // Optional context — a failed read must never block the widget.
881
+ return null
882
+ }
883
+ }
884
+
885
+ // The discovery sequence for one endpoint's reference resource, kept here
886
+ // rather than in the panel so the decision AND the call site are testable
887
+ // together: a gate nothing consults is exactly the bug this shape prevents.
888
+ //
889
+ // A volatile resource is NOT read here. Its content is only meaningful for
890
+ // the turn it is attached to, so reading it at boot would buy a copy that is
891
+ // already suspect by the time anyone sends a message.
866
892
  export async function loadHostResource({ endpoint, budgetBytes, list, read, onSkip }) {
867
893
  const resources = await list({ endpoint })
868
894
  const candidate = (resources || []).filter((r) => (r.mimeType || "") === "application/json")[0]
869
895
  if (!candidate) return null
870
896
 
871
897
  const plan = planResourceAttachment(candidate, budgetBytes)
898
+ plan.name = candidate.name || candidate.uri
899
+ // Carried so a volatile re-read knows where to go: discovery happens once
900
+ // at boot, but the fetch it authorises happens on every later turn.
901
+ plan.endpoint = endpoint
872
902
  if (!plan.fetch) {
873
903
  if (onSkip) onSkip(plan)
874
904
  return { plan, context: null }
875
905
  }
906
+ if (plan.volatility === "volatile") return { plan, context: null }
876
907
 
877
- try {
878
- const result = await read({ endpoint, uri: candidate.uri })
879
- const entry = ((result && result.contents) || [])[0]
880
- if (!entry || !entry.text) return { plan, context: null }
881
- return {
882
- plan,
883
- context: {
884
- uri: candidate.uri,
885
- name: candidate.name || candidate.uri,
886
- text: trimResourceText(entry.text, budgetBytes)
887
- }
888
- }
889
- } catch (e) {
890
- // Optional context — a failed read must never block the widget.
891
- return { plan, context: null }
908
+ const context = await readResourceContext({
909
+ endpoint, uri: candidate.uri, name: plan.name, read, budgetBytes
910
+ })
911
+ return { plan, context }
912
+ }
913
+
914
+ export function resourceContextLines(context) {
915
+ if (!context) return []
916
+ return [ "", "Reference data — " + context.name + " (" + context.uri + "):", context.text ]
917
+ }
918
+
919
+ // What to attach on THIS turn.
920
+ //
921
+ // `volatility` governs RE-FETCHING, not re-inclusion: a stable resource is
922
+ // read once and then re-used, but it stays in every turn's system prompt for
923
+ // as long as the conversation lasts. Showing it only on the first turn put
924
+ // the reference data in the one turn that could not use it — the model then
925
+ // spent several tool calls rediscovering what it had already been given,
926
+ // which costs more tokens than simply keeping it.
927
+ //
928
+ // Whether a resource is affordable at all is decided once, from sizeBytes,
929
+ // before it is ever fetched.
930
+ export async function resourceLinesForTurn({ plan, cached, endpoint, read, budgetBytes }) {
931
+ if (!plan || !plan.autoAttach) return { lines: [], cached }
932
+
933
+ if (plan.volatility === "volatile") {
934
+ const fresh = await readResourceContext({
935
+ endpoint, uri: plan.uri, name: plan.name, read, budgetBytes
936
+ })
937
+ return fresh ? { lines: resourceContextLines(fresh), cached: fresh } : { lines: [], cached }
892
938
  }
939
+
940
+ return { lines: resourceContextLines(cached), cached }
893
941
  }
894
942
 
895
943
  // ---- prompt templates ---------------------------------------------------
@@ -930,6 +978,16 @@ export function resolvePromptArguments(prompt, state) {
930
978
  return { args, missing }
931
979
  }
932
980
 
981
+ // What a template will take from the page, and what it will have to ask for.
982
+ // Shown to a first-time visitor so the offer is concrete rather than a bare
983
+ // verb: they can see that the text box is empty before they click.
984
+ export function promptArgumentSummary(prompt, state) {
985
+ return ((prompt && prompt.arguments) || []).map((arg) => {
986
+ const value = promptArgFromState(arg.name, state)
987
+ return { name: arg.name, value, filled: value.length > 0, required: !!arg.required }
988
+ })
989
+ }
990
+
933
991
  export function promptButtonProps(prompt) {
934
992
  return {
935
993
  label: prompt.title || prompt.name,
@@ -937,9 +995,3 @@ export function promptButtonProps(prompt) {
937
995
  }
938
996
  }
939
997
 
940
- // The per-turn attach decision AND its formatting, so the two cannot drift
941
- // apart: asking whether to attach is what consumes the turn.
942
- export function nextResourceContextLines(context, attacher) {
943
- if (!context || !attacher || !attacher.take()) return []
944
- return [ "", "Reference data — " + context.name + " (" + context.uri + "):", context.text ]
945
- }
@@ -181,3 +181,51 @@
181
181
  max-height: 10em;
182
182
  overflow-y: auto;
183
183
  }
184
+
185
+ /* ---- "working" signals, ported from llm_meta_chat's chats.css ------------
186
+ *
187
+ * The wait is often long on local models — a tool round re-processes the
188
+ * whole prompt plus the tool results before the first token — so a still
189
+ * label reads as a hang. Same class names as the chat app, so the two
190
+ * surfaces behave identically. */
191
+ .message-role .role-spinner {
192
+ display: inline-block;
193
+ animation: role-working-spin 1.8s linear infinite;
194
+ transform-origin: 50% 50%;
195
+ }
196
+
197
+ @keyframes role-working-spin {
198
+ from { transform: rotate(0deg); }
199
+ to { transform: rotate(360deg); }
200
+ }
201
+
202
+ /* Staggered three-dot indicator on the reasoning summary while thinking is
203
+ * still streaming. Hidden unless .thinking-active is present. */
204
+ .thinking-dots {
205
+ display: none;
206
+ margin-left: 0.35em;
207
+ letter-spacing: 0.15em;
208
+ font-weight: bold;
209
+ }
210
+
211
+ .message-thinking.thinking-active .thinking-dots { display: inline-block; }
212
+
213
+ .message-thinking.thinking-active .thinking-dots span {
214
+ display: inline-block;
215
+ opacity: 0.25;
216
+ animation: thinking-dot 1.4s ease-in-out infinite;
217
+ }
218
+
219
+ .message-thinking.thinking-active .thinking-dots span:nth-child(2) { animation-delay: 0.2s; }
220
+ .message-thinking.thinking-active .thinking-dots span:nth-child(3) { animation-delay: 0.4s; }
221
+
222
+ @keyframes thinking-dot {
223
+ 0%, 60%, 100% { opacity: 0.25; }
224
+ 30% { opacity: 1; }
225
+ }
226
+
227
+ /* Respect a reader's motion preference: keep the label, drop the movement. */
228
+ @media (prefers-reduced-motion: reduce) {
229
+ .message-role .role-spinner,
230
+ .message-thinking.thinking-active .thinking-dots span { animation: none; }
231
+ }
@@ -35,7 +35,11 @@ module LlmMetaWidget
35
35
  # anon" (all Ollama models / all public_to_anonymous MCP servers).
36
36
  # Pass arrays to curate.
37
37
  models: nil, # e.g. ["qwen3-6-35b-fast", "qwen3-6-35b-no-think"]
38
- hub_tools: nil # e.g. ["togomcp", "pubdictionaries"] — MCP server names
38
+ hub_tools: nil, # e.g. ["togomcp", "pubdictionaries"] — MCP server names
39
+ # First thing a visitor sees when the panel opens, above the offered
40
+ # prompt templates. nil → a generic line. A blank panel tells a
41
+ # first-time visitor nothing about what the assistant can do for them.
42
+ greeting: nil
39
43
  }.freeze
40
44
 
41
45
  def llm_meta_widget(base_url:, model:, **overrides)
@@ -268,6 +268,59 @@
268
268
  border-radius: 8px;
269
269
  padding: 8px;
270
270
  }
271
+ #llm-meta-widget-chat .message-content details > summary {
272
+ cursor: pointer;
273
+ font-weight: 600;
274
+ }
275
+ #llm-meta-widget-chat .lmw-sent-text {
276
+ margin-top: 6px;
277
+ font-size: 12px;
278
+ opacity: 0.8;
279
+ white-space: pre-wrap;
280
+ }
281
+
282
+ /* Opening state: greeting + the offers themselves. Shown until the first
283
+ * real turn, and again after Clear. */
284
+ #llm-meta-widget-chat .lmw-welcome { padding: 4px 2px 2px; }
285
+ #llm-meta-widget-chat .lmw-welcome-hello {
286
+ margin: 0 0 12px;
287
+ font-size: 14px;
288
+ line-height: 1.5;
289
+ color: #374151;
290
+ }
291
+ #llm-meta-widget-chat .lmw-welcome-card {
292
+ background-color: #f9fafb;
293
+ border: 1px solid #e5e7eb;
294
+ border-radius: 8px;
295
+ padding: 10px 12px;
296
+ margin-bottom: 8px;
297
+ }
298
+ #llm-meta-widget-chat .lmw-welcome-start {
299
+ background-color: #eff6ff;
300
+ color: #1d4ed8;
301
+ border: 1px solid #bfdbfe;
302
+ border-radius: 999px;
303
+ padding: 4px 12px;
304
+ font-size: 13px;
305
+ font-family: inherit;
306
+ cursor: pointer;
307
+ }
308
+ #llm-meta-widget-chat .lmw-welcome-start:hover { background-color: #dbeafe; }
309
+ #llm-meta-widget-chat .lmw-welcome-why {
310
+ margin: 8px 0 0;
311
+ font-size: 13px;
312
+ line-height: 1.45;
313
+ color: #4b5563;
314
+ }
315
+ #llm-meta-widget-chat .lmw-welcome-slots {
316
+ margin: 8px 0 0;
317
+ padding-left: 16px;
318
+ font-size: 12px;
319
+ color: #6b7280;
320
+ }
321
+ #llm-meta-widget-chat .lmw-welcome-slots li { margin-bottom: 2px; }
322
+ #llm-meta-widget-chat .lmw-welcome-slots li.filled { color: #047857; }
323
+
271
324
  /* Server-offered prompt templates. The row is display:none in markup and
272
325
  * gets its display back only when a prompt actually exists, so the inline
273
326
  * style and this rule have to agree on "flex". */
@@ -398,8 +451,8 @@
398
451
  <script type="module">
399
452
  import { runChatLoop, fetchMcpManifest, listMcpPrompts, getMcpPrompt,
400
453
  listMcpResources, readMcpResource, promptMessagesToText,
401
- loadHostResource, createResourceAttacher, resolvePromptArguments,
402
- promptButtonProps, nextResourceContextLines } from "<%= orchestrator_path %>";
454
+ loadHostResource, resourceLinesForTurn, resolvePromptArguments,
455
+ promptButtonProps, promptArgumentSummary } from "<%= orchestrator_path %>";
403
456
  import { marked } from "/llm_meta_widget_assets/marked.esm.js";
404
457
 
405
458
  // Standard prose settings — GFM (tables, autolinks, strikethrough),
@@ -411,6 +464,7 @@
411
464
  var API_KEY_UUID = <%= api_key_uuid.to_json.html_safe %>;
412
465
  var MODEL = <%= model.to_json.html_safe %>;
413
466
  var ACTIONS_SCHEMA_ID = <%= actions_schema_id.to_json.html_safe %>;
467
+ var GREETING = <%= greeting.to_json.html_safe %>;
414
468
  var STATE_GLOBAL = <%= state_global.to_json.html_safe %>;
415
469
  var ACTIONS_GLOBAL = <%= actions_global.to_json.html_safe %>;
416
470
  var REMOTE_TOOLS_SCHEMA_ID = <%= remote_tools_schema_id.to_json.html_safe %>;
@@ -755,7 +809,6 @@
755
809
  var hostWideTools = [];
756
810
  var hostWidePrompts = [];
757
811
  var resourceContext = null; // payload of the host's reference resource
758
- var resourceAttacher = null; // per-turn gate, built from the server's hint
759
812
  var resourcePlan = null; // the pre-flight decision, kept so Clear can re-arm the gate
760
813
 
761
814
  // Beyond ~2k tokens a reference resource starts crowding out the
@@ -763,9 +816,6 @@
763
816
  // is 218 dictionaries / ~35KB / ~10k tokens.
764
817
  var RESOURCE_BUDGET_BYTES = 8000;
765
818
 
766
- // Set when a server declared a resource we deliberately did not attach,
767
- // so the reason is inspectable rather than a silent absence.
768
- var resourceSkip = null;
769
819
 
770
820
  var wellKnownReady = (async function() {
771
821
  var urls = WELL_KNOWN_URLS === null
@@ -789,7 +839,7 @@
789
839
  var prompts = await listMcpPrompts({ endpoint: endpoint });
790
840
  hostWidePrompts = hostWidePrompts.concat(prompts);
791
841
 
792
- if (resourceContext === null) {
842
+ if (resourcePlan === null) {
793
843
  // Listing, size gate and read all live in the orchestrator, where
794
844
  // they are tested together — a gate nothing consults is exactly
795
845
  // the bug this shape prevents.
@@ -799,19 +849,21 @@
799
849
  list: listMcpResources,
800
850
  read: readMcpResource,
801
851
  onSkip: function(plan) {
802
- resourceSkip = plan;
803
852
  console.info("[llm_meta_widget] not attaching " + plan.uri +
804
853
  " (" + plan.reason + ", sizeBytes=" + plan.sizeBytes + ")");
805
854
  }
806
855
  });
807
- if (loaded && loaded.context) {
808
- resourceContext = loaded.context;
809
- resourcePlan = loaded.plan;
810
- resourceAttacher = createResourceAttacher(loaded.plan);
856
+ // A volatile resource comes back with no context — it is read
857
+ // per turn instead — so the plan, not the payload, is what says
858
+ // whether this endpoint offered anything worth attaching.
859
+ if (loaded && loaded.plan && loaded.plan.fetch) {
860
+ resourceContext = loaded.context;
861
+ resourcePlan = loaded.plan;
811
862
  }
812
863
  }
813
864
  }
814
865
  renderPromptButtons();
866
+ renderWelcome();
815
867
  })();
816
868
 
817
869
  // ---- server-offered prompt templates --------------------------------
@@ -841,6 +893,66 @@
841
893
  promptsEl.style.display = "";
842
894
  }
843
895
 
896
+ // A blank panel tells a first-time visitor nothing. Open with a greeting
897
+ // and the offers themselves — each template showing what it will take from
898
+ // the page and what it will ask for — so the assistant is the page's way
899
+ // in rather than a box you must already know how to talk to.
900
+ function renderWelcome() {
901
+ if (!historyEl || historyEl.querySelector(".message")) return;
902
+ historyEl.textContent = "";
903
+
904
+ var box = document.createElement("div");
905
+ box.className = "lmw-welcome";
906
+
907
+ var hello = document.createElement("p");
908
+ hello.className = "lmw-welcome-hello";
909
+ hello.textContent = GREETING ||
910
+ "Hi — I can work this page for you. Tell me what you need in your own words" +
911
+ (hostWidePrompts.length ? ", or start with one of these:" : ".");
912
+ box.appendChild(hello);
913
+
914
+ var state = window[STATE_GLOBAL] || {};
915
+ hostWidePrompts.forEach(function(prompt) {
916
+ var props = promptButtonProps(prompt);
917
+ var card = document.createElement("div");
918
+ card.className = "lmw-welcome-card";
919
+
920
+ var start = document.createElement("button");
921
+ start.type = "button";
922
+ start.className = "lmw-welcome-start";
923
+ start.textContent = props.label;
924
+ start.addEventListener("click", function() { runPromptTemplate(prompt, start); });
925
+ card.appendChild(start);
926
+
927
+ if (prompt.description) {
928
+ var why = document.createElement("p");
929
+ why.className = "lmw-welcome-why";
930
+ why.textContent = prompt.description;
931
+ card.appendChild(why);
932
+ }
933
+
934
+ // Name the placeholders, filled or not, so the offer is concrete:
935
+ // a newcomer can see the text box is empty before clicking.
936
+ var summary = promptArgumentSummary(prompt, state);
937
+ if (summary.length) {
938
+ var slots = document.createElement("ul");
939
+ slots.className = "lmw-welcome-slots";
940
+ summary.forEach(function(slot) {
941
+ var li = document.createElement("li");
942
+ li.className = slot.filled ? "filled" : "empty";
943
+ li.textContent = slot.filled
944
+ ? slot.name + ": " + (slot.value.length > 60 ? slot.value.slice(0, 60) + "…" : slot.value)
945
+ : slot.name + ": not set yet — I'll ask, or work it out";
946
+ slots.appendChild(li);
947
+ });
948
+ card.appendChild(slots);
949
+ }
950
+ box.appendChild(card);
951
+ });
952
+
953
+ historyEl.appendChild(box);
954
+ }
955
+
844
956
  async function runPromptTemplate(prompt, button) {
845
957
  var resolved = resolvePromptArguments(prompt, window[STATE_GLOBAL] || {});
846
958
  var args = resolved.args;
@@ -860,6 +972,7 @@
860
972
  var text = promptMessagesToText(result);
861
973
  if (!text) { appendTurn("error", "The server returned an empty prompt."); return; }
862
974
  inputEl.value = text;
975
+ pendingTurnLabel = promptButtonProps(prompt).label;
863
976
  if (typeof formEl.requestSubmit === "function") formEl.requestSubmit();
864
977
  else formEl.dispatchEvent(new Event("submit", { cancelable: true }));
865
978
  } catch (e) {
@@ -869,13 +982,22 @@
869
982
  }
870
983
  }
871
984
 
872
- function appendTurn(role, text) {
985
+ // Set just before a template submits, so its turn is labelled by what the
986
+ // user actually did — "Annotate text" — instead of showing a paragraph of
987
+ // server-written instructions as though they had typed it. The instructions
988
+ // stay one click away rather than hidden: what was sent is what is shown.
989
+ var pendingTurnLabel = null;
990
+
991
+ function appendTurn(role, text, turnLabel) {
873
992
  // Class names mirror llm_meta_chat's chats/_message.html.erb —
874
993
  // `.message.<role>`, `.message-role`, `.message-content` — so the
875
994
  // shared conversation.css styles apply directly (see the <link>
876
995
  // above). The .lmw-* prefix is reserved for widget-CHROME classes
877
996
  // (header, clear button, scroll region, input area) that aren't
878
997
  // part of the shared conversation surface.
998
+ var welcome = historyEl.querySelector(".lmw-welcome");
999
+ if (welcome) welcome.remove();
1000
+
879
1001
  var div = document.createElement("div");
880
1002
  div.className = "message " + role;
881
1003
  var label = document.createElement("div");
@@ -888,12 +1010,25 @@
888
1010
  // content is rendered as markdown but only after the assistant's
889
1011
  // text is streamed in via renderMarkdownInto — this appendTurn
890
1012
  // creates the empty container.
891
- body.textContent = text;
1013
+ if (turnLabel) {
1014
+ var details = document.createElement("details");
1015
+ var summary = document.createElement("summary");
1016
+ summary.textContent = turnLabel;
1017
+ var full = document.createElement("div");
1018
+ full.className = "lmw-sent-text";
1019
+ full.textContent = text;
1020
+ details.appendChild(summary);
1021
+ details.appendChild(full);
1022
+ body.appendChild(details);
1023
+ } else {
1024
+ body.textContent = text;
1025
+ }
892
1026
  div.appendChild(label);
893
1027
  div.appendChild(document.createTextNode(" "));
894
1028
  div.appendChild(body);
895
1029
  historyEl.appendChild(div);
896
1030
  historyEl.scrollTop = historyEl.scrollHeight;
1031
+ body.roleLabel = label; // so a turn in flight can show it is working
897
1032
  return body;
898
1033
  }
899
1034
 
@@ -918,6 +1053,23 @@
918
1053
  return role;
919
1054
  }
920
1055
 
1056
+ // While a turn is in flight the assistant's label becomes a turning gear.
1057
+ // Same markup and class names as llm_meta_chat's message_stream_controller,
1058
+ // so the shared conversation.css drives both. Without it there is no sign
1059
+ // whether the assistant is still working or has quietly stopped — and on a
1060
+ // local model a tool round can take a long time before the first token.
1061
+ function markWorking(label) {
1062
+ if (!label || label.classList.contains("is-working")) return;
1063
+ label.innerHTML = '<span class="role-spinner" aria-hidden="true">\u2699\uFE0F</span> Working…';
1064
+ label.classList.add("is-working");
1065
+ }
1066
+
1067
+ function markDone(label) {
1068
+ if (!label || !label.classList.contains("is-working")) return;
1069
+ label.classList.remove("is-working");
1070
+ label.textContent = roleLabel("assistant");
1071
+ }
1072
+
921
1073
  var currentThinkingBlock = null;
922
1074
  var currentThinkingBody = null;
923
1075
 
@@ -928,8 +1080,17 @@
928
1080
  var details = document.createElement("details");
929
1081
  details.className = "message-thinking";
930
1082
  details.open = true;
1083
+ details.classList.add("thinking-active");
931
1084
  var summary = document.createElement("summary");
932
1085
  summary.textContent = "🤔 thinking…";
1086
+ var dots = document.createElement("span");
1087
+ dots.className = "thinking-dots";
1088
+ for (var i = 0; i < 3; i++) {
1089
+ var dot = document.createElement("span");
1090
+ dot.textContent = ".";
1091
+ dots.appendChild(dot);
1092
+ }
1093
+ summary.appendChild(dots);
933
1094
  var body = document.createElement("div");
934
1095
  body.className = "message-thinking-content";
935
1096
  details.appendChild(summary);
@@ -943,6 +1104,7 @@
943
1104
 
944
1105
  function collapseThinkingBlock() {
945
1106
  if (currentThinkingBlock) {
1107
+ currentThinkingBlock.classList.remove("thinking-active");
946
1108
  currentThinkingBlock.open = false;
947
1109
  var summary = currentThinkingBlock.querySelector("summary");
948
1110
  if (summary) summary.textContent = "🤔 thinking (finished)";
@@ -960,28 +1122,35 @@
960
1122
  return out;
961
1123
  }
962
1124
 
963
- function currentSystemPrompt() {
1125
+ function currentSystemPrompt(resourceLines) {
964
1126
  return [
965
1127
  "You are integrated into a web page as an AI assistant. You have tools available to change page state or fetch information.",
966
1128
  "",
967
1129
  "RULES for tool use:",
968
1130
  "1. If the user's question can be answered from the Current page state below, answer directly with a plain-text response — do NOT invoke a tool.",
969
1131
  "2. If the user requests a state change, or needs information not in the page state, invoke the matching tool via a function call. Do not describe your intent in text without actually invoking (a textual promise like \"I will add X\" is a failure).",
970
- "3. After a tool returns a result, YOUR NEXT MESSAGE MUST BE A PLAIN-TEXT ANSWER using that result. DO NOT emit another tool call.",
971
- "4. NEVER call the SAME tool twice in a row with the same or similar arguments — its earlier result is already in the conversation history.",
1132
+ "3. After a tool returns a result, use it: either take the next step the task needs, or — if the task is done — answer in plain text. Do not stop silently after a tool call.",
1133
+ "4. NEVER repeat a call you have already made with the same or similar arguments — its result is already in the conversation history.",
972
1134
  "",
973
1135
  "Current page state:",
974
1136
  JSON.stringify(currentPageState(), null, 2)
975
- ].concat(resourceContextLines()).join("\n");
1137
+ ].concat(resourceLines || []).join("\n");
976
1138
  }
977
1139
 
978
- // How often the resource is attached is the SERVER's call, via the
979
- // extension's attachmentHint: 'once' for static reference data (the
980
- // default, and what a hint-unaware server gets), 'each-turn' for content
981
- // that varies. The attacher records each turn's answer, replacing a
982
- // client-side boolean that assumed every resource was static.
983
- function resourceContextLines() {
984
- return nextResourceContextLines(resourceContext, resourceAttacher);
1140
+ // Runs once per send, before the system prompt is built: a volatile
1141
+ // resource is re-read here, a stable one re-uses the copy taken at boot.
1142
+ // Either way it is attached to every turn — the server's hint governs
1143
+ // re-FETCHING; how often to include it is this client's call.
1144
+ async function resourceLinesForThisTurn() {
1145
+ var turn = await resourceLinesForTurn({
1146
+ plan: resourcePlan,
1147
+ cached: resourceContext,
1148
+ endpoint: resourcePlan && resourcePlan.endpoint,
1149
+ read: readMcpResource,
1150
+ budgetBytes: RESOURCE_BUDGET_BYTES
1151
+ });
1152
+ resourceContext = turn.cached;
1153
+ return turn.lines;
985
1154
  }
986
1155
 
987
1156
  // Per-turn AbortController — lets the Clear button (or a new submit)
@@ -992,9 +1161,7 @@
992
1161
  if (currentAbort) { try { currentAbort.abort(); } catch (e) { /* noop */ } }
993
1162
  conversation = [];
994
1163
  historyEl.innerHTML = "";
995
- // Clear starts a new conversation, so a 'once' resource is owed to it
996
- // again — the old boolean stayed latched and silently withheld it.
997
- if (resourcePlan) resourceAttacher = createResourceAttacher(resourcePlan);
1164
+ renderWelcome();
998
1165
  });
999
1166
 
1000
1167
  // Enter submits, Shift+Enter inserts a newline — matches llm_meta_chat's
@@ -1008,17 +1175,19 @@
1008
1175
  }
1009
1176
  });
1010
1177
 
1011
- formEl.addEventListener("submit", async function(e) {
1012
- e.preventDefault();
1178
+ formEl.addEventListener("submit", async function(event) {
1179
+ event.preventDefault();
1013
1180
  var userText = inputEl.value.trim();
1014
1181
  if (!userText) return;
1015
1182
  inputEl.value = "";
1016
- appendTurn("user", userText);
1183
+ appendTurn("user", userText, pendingTurnLabel);
1184
+ pendingTurnLabel = null;
1017
1185
 
1018
1186
  if (currentAbort) { try { currentAbort.abort(); } catch (e) { /* noop */ } }
1019
1187
  currentAbort = new AbortController();
1020
1188
 
1021
1189
  var assistantBody = appendTurn("assistant", "");
1190
+ markWorking(assistantBody.roleLabel);
1022
1191
  var assistantMarkdown = ""; // accumulate raw markdown, re-render on each delta
1023
1192
 
1024
1193
  try {
@@ -1028,7 +1197,8 @@
1028
1197
  // it as sent.
1029
1198
  await wellKnownReady;
1030
1199
 
1031
- var messages = [{ role: "system", content: currentSystemPrompt() }]
1200
+ var resourceLines = await resourceLinesForThisTurn();
1201
+ var messages = [{ role: "system", content: currentSystemPrompt(resourceLines) }]
1032
1202
  .concat(conversation)
1033
1203
  .concat([{ role: "user", content: userText }]);
1034
1204
 
@@ -1043,7 +1213,16 @@
1043
1213
  aiActions: window[ACTIONS_GLOBAL] || {},
1044
1214
  maxRounds: MAX_ROUNDS,
1045
1215
  signal: currentAbort.signal,
1216
+ onPhase: function(name) {
1217
+ // 'thinking' covers the long silence before the first
1218
+ // token; 'responding' means text is on its way.
1219
+ if (name === "responding") markDone(assistantBody.roleLabel);
1220
+ else markWorking(assistantBody.roleLabel);
1221
+ },
1046
1222
  onRoundStart: function(roundIdx) {
1223
+ // A new round means more work: tool results are going back
1224
+ // to the model, which is the longest wait of all.
1225
+ markWorking(assistantBody.roleLabel);
1047
1226
  // Loop mechanics are debugging info, not user-facing signal.
1048
1227
  // Reuse the same assistant bubble across rounds — text just
1049
1228
  // keeps streaming into it (accumulating markdown). Weaker
@@ -1060,6 +1239,7 @@
1060
1239
  historyEl.scrollTop = historyEl.scrollHeight;
1061
1240
  },
1062
1241
  onTextDelta: function(delta) {
1242
+ markDone(assistantBody.roleLabel);
1063
1243
  collapseThinkingBlock();
1064
1244
  assistantMarkdown += delta;
1065
1245
  renderMarkdownInto(assistantBody, assistantMarkdown);
@@ -1110,6 +1290,12 @@
1110
1290
  appendTurn("error", err.message);
1111
1291
  }
1112
1292
  } finally {
1293
+ // Whatever happened — answered, aborted, threw, or returned
1294
+ // nothing at all — the turn is over and the label must stop
1295
+ // claiming otherwise. A spinner left running is a worse lie than
1296
+ // no spinner: it says "still working" about a turn that ended.
1297
+ markDone(assistantBody.roleLabel);
1298
+ collapseThinkingBlock();
1113
1299
  currentAbort = null;
1114
1300
  }
1115
1301
  });
@@ -1,3 +1,3 @@
1
1
  module LlmMetaWidget
2
- VERSION = "0.2.0"
2
+ VERSION = "0.4.0"
3
3
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: llm_meta_widget
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.2.0
4
+ version: 0.4.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - jdkim