ai-sdk-ollama 4.2.0 → 4.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/CHANGELOG.md +23 -0
  2. package/README.md +78 -13
  3. package/dist/functions/generate-text.d.ts.map +1 -1
  4. package/dist/functions/stream-text.d.ts.map +1 -1
  5. package/dist/index.browser.cjs +7558 -2236
  6. package/dist/index.browser.cjs.map +1 -1
  7. package/dist/index.browser.d.ts +2 -0
  8. package/dist/index.browser.d.ts.map +1 -1
  9. package/dist/index.browser.js +7558 -2236
  10. package/dist/index.browser.js.map +1 -1
  11. package/dist/index.cjs +7591 -2266
  12. package/dist/index.cjs.map +1 -1
  13. package/dist/index.d.ts +2 -0
  14. package/dist/index.d.ts.map +1 -1
  15. package/dist/index.js +7590 -2266
  16. package/dist/index.js.map +1 -1
  17. package/dist/models/chat-language-model.d.ts +9 -2
  18. package/dist/models/chat-language-model.d.ts.map +1 -1
  19. package/dist/models/chat-language-model.test-helpers.d.ts +3 -1
  20. package/dist/models/chat-language-model.test-helpers.d.ts.map +1 -1
  21. package/dist/models/chat-result.d.ts +1 -2
  22. package/dist/models/chat-result.d.ts.map +1 -1
  23. package/dist/models/chat-stream.d.ts +2 -3
  24. package/dist/models/chat-stream.d.ts.map +1 -1
  25. package/dist/models/embedding-model.d.ts +2 -2
  26. package/dist/models/embedding-model.d.ts.map +1 -1
  27. package/dist/models/embedding-reranking-model.d.ts +3 -3
  28. package/dist/models/embedding-reranking-model.d.ts.map +1 -1
  29. package/dist/models/evaluation-model.d.ts +21 -0
  30. package/dist/models/evaluation-model.d.ts.map +1 -0
  31. package/dist/ollama-client.d.ts +42 -0
  32. package/dist/ollama-client.d.ts.map +1 -0
  33. package/dist/provider.browser.d.ts.map +1 -1
  34. package/dist/provider.d.ts +11 -10
  35. package/dist/provider.d.ts.map +1 -1
  36. package/dist/tool/web-fetch.d.ts +6 -4
  37. package/dist/tool/web-fetch.d.ts.map +1 -1
  38. package/dist/tool/web-search.d.ts +8 -6
  39. package/dist/tool/web-search.d.ts.map +1 -1
  40. package/dist/utils/json-schema-coercion.d.ts +3 -4
  41. package/dist/utils/json-schema-coercion.d.ts.map +1 -1
  42. package/dist/utils/json-text-repair.d.ts +9 -6
  43. package/dist/utils/json-text-repair.d.ts.map +1 -1
  44. package/dist/utils/last-user-text.d.ts +7 -0
  45. package/dist/utils/last-user-text.d.ts.map +1 -0
  46. package/dist/utils/object-generation-reliability.d.ts +3 -4
  47. package/dist/utils/object-generation-reliability.d.ts.map +1 -1
  48. package/dist/utils/ollama-error.d.ts +2 -0
  49. package/dist/utils/ollama-error.d.ts.map +1 -1
  50. package/package.json +18 -21
package/CHANGELOG.md CHANGED
@@ -1,5 +1,28 @@
1
1
  # Changelog
2
2
 
3
+ ## 4.4.0
4
+
5
+ ### Minor Changes
6
+
7
+ - 004f6b4: Add `ollama.evaluationModel()` for Ollama decision models such as `nimble` and `tev1`. Use it with the AI SDK's `experimental_evaluate` to get typed choice, boolean and score answers with probabilities from one local call.
8
+
9
+ - `OllamaClient` accepts an optional `systemone()` method.
10
+ - `generateText` and `streamText` restate the latest user message, text parts included, when they synthesise a reply from tool results.
11
+ - JSON repair serialises object values for string fields as JSON.
12
+ - Supports `ai` 7.0.103+ and `ollama` 0.6.4.
13
+
14
+ ## 4.3.0
15
+
16
+ ### Minor Changes
17
+
18
+ - 371f515: Track AI SDK 7.0.95 and open the provider up to custom Ollama clients.
19
+
20
+ - The `ai` peer range moves to `^7.0.95`, with `@ai-sdk/provider` and `@ai-sdk/provider-utils` updated to match.
21
+ - `OllamaClient` is a structural contract, so the official client or a custom adapter can be injected in both Node.js and browser builds. Streaming methods use the newly exported `AbortableStream` type.
22
+ - Per-request `AbortSignal`s are forwarded to the client, and an aborted request settles as a rejection across the reliability retry and fallback paths.
23
+ - Returned `message.thinking` is surfaced as AI SDK reasoning independently of the request's `think` setting, including on terminal stream chunks, and streamed text and reasoning blocks close symmetrically.
24
+ - Output usage carries the total token count Ollama reports and leaves the text/reasoning split unset, matching what the API provides.
25
+
3
26
  ## 4.2.0
4
27
 
5
28
  ### Minor Changes
package/README.md CHANGED
@@ -154,6 +154,7 @@ export OLLAMA_API_KEY="your_api_key_here"
154
154
  - [Combining Tools with Structured Output](#combining-tools-with-structured-output)
155
155
  - [Simple and Predictable](#simple-and-predictable)
156
156
  - [Reranking](#reranking)
157
+ - [Decision Models](#decision-models)
157
158
  - [Streaming Utilities](#streaming-utilities)
158
159
  - [Smooth Stream](#smooth-stream)
159
160
  - [Partial JSON Parsing](#partial-json-parsing)
@@ -481,6 +482,46 @@ ranking.forEach((item, i) => {
481
482
 
482
483
  **Recommended Models**: `embeddinggemma` (best score separation), `nomic-embed-text`, `bge-m3`
483
484
 
485
+ ## Decision Models
486
+
487
+ Ollama 0.35 serves decision models (`nimble`, `tev1`, `tev1:0.8b`) on `/v1/systemone`. You ask typed questions about some state and get each answer with its probabilities in one local call. Use them for triage, routing and moderation through the AI SDK's `experimental_evaluate`:
488
+
489
+ ```typescript
490
+ import { experimental_evaluate as evaluate } from 'ai';
491
+ import { ollama } from 'ai-sdk-ollama';
492
+
493
+ const { answers } = await evaluate({
494
+ model: ollama.evaluationModel('nimble'),
495
+ state: { ticket: 'I was charged twice. Please refund the extra payment.' },
496
+ questions: {
497
+ team: {
498
+ type: 'choice',
499
+ instructions: 'Which team should handle this ticket?',
500
+ criteria: {
501
+ billing: 'Payments and refunds',
502
+ technical: null,
503
+ other: null,
504
+ },
505
+ },
506
+ refund: {
507
+ type: 'boolean',
508
+ instructions: 'Does the customer ask for a refund?',
509
+ },
510
+ urgency: {
511
+ type: 'score',
512
+ instructions: 'How urgent is this ticket?',
513
+ criteria: ['Routine', 'Soon', 'Urgent'],
514
+ },
515
+ },
516
+ });
517
+
518
+ answers.team.choice; // 'billing'
519
+ answers.refund.probability; // P(true), e.g. 0.99
520
+ answers.urgency.score; // 0 to 2, e.g. 0.8
521
+ ```
522
+
523
+ Ollama names boolean questions `noul`, and the provider maps them for you. Ollama's per-question `confidence` lands in `providerMetadata.ollama.confidence`. Pull a model first with `ollama pull nimble`.
524
+
484
525
  ## Streaming Utilities
485
526
 
486
527
  ### Smooth Stream
@@ -740,7 +781,9 @@ const ollama = createOllama({ apiKey: process.env.OLLAMA_API_KEY });
740
781
 
741
782
  ### Using Existing Ollama Client
742
783
 
743
- You can also pass an existing Ollama client instance to reuse your configuration:
784
+ You can also pass an existing Ollama client instance to reuse your configuration.
785
+ The provider accepts the exported structural `OllamaClient` interface, so you
786
+ can inject the official client or a custom adapter in Node.js and browser builds.
744
787
 
745
788
  ```typescript
746
789
  import { Ollama } from 'ollama';
@@ -756,13 +799,36 @@ const existingClient = new Ollama({
756
799
  const ollamaSdk = createOllama({ client: existingClient });
757
800
 
758
801
  // Use both clients as needed
759
- await ollamaRaw.list(); // Direct Ollama operations
802
+ await existingClient.list(); // Direct Ollama operations
760
803
  const { text } = await generateText({
761
804
  model: ollamaSdk('llama3.2'),
762
805
  prompt: 'Hello!',
763
806
  });
764
807
  ```
765
808
 
809
+ #### Custom client adapters
810
+
811
+ `OllamaClient` is structural, so any object with the right shape works — no
812
+ `Ollama` instance required. Streaming methods return an `AbortableStream`, an
813
+ async iterable with an `abort()`:
814
+
815
+ ```typescript
816
+ import type { AbortableStream, OllamaClient } from 'ai-sdk-ollama';
817
+ import { createOllama } from 'ai-sdk-ollama';
818
+
819
+ const client: OllamaClient = {
820
+ chat: myChat, // returns a ChatResponse, or an AbortableStream of them
821
+ embed: myEmbed,
822
+ webSearch: myWebSearch,
823
+ webFetch: myWebFetch,
824
+ };
825
+
826
+ const ollama = createOllama({ client });
827
+ ```
828
+
829
+ Per-request cancellation is forwarded to the client as an `AbortSignal`, and an
830
+ aborted request rejects rather than resolving with a retried or fallback result.
831
+
766
832
  ### Structured Output
767
833
 
768
834
  ```typescript
@@ -918,39 +984,38 @@ const { output: custom } = await generateText({
918
984
 
919
985
  ### Reasoning Support
920
986
 
921
- Some models like DeepSeek-R1 support reasoning (chain-of-thought) output. Enable this feature to see the model's thinking process:
987
+ Use `think` to request reasoning generation from models that support it. Returned `message.thinking` is exposed as AI SDK reasoning, separately from final text, in both generation results and streams (including terminal chunks).
922
988
 
923
989
  ```typescript
924
990
  import { ollama } from 'ai-sdk-ollama';
925
991
  import { generateText } from 'ai';
926
992
 
927
993
  // Enable reasoning for models that support it (e.g., deepseek-r1:7b)
928
- const model = ollama('deepseek-r1:7b', { reasoning: true });
994
+ const model = ollama('deepseek-r1:7b', { think: true });
929
995
 
930
996
  // Generate text with reasoning
931
- const { text } = await generateText({
997
+ const { text, reasoningText } = await generateText({
932
998
  model,
933
999
  prompt:
934
1000
  'Solve: If I have 3 boxes, each with 4 smaller boxes, and each smaller box has 5 items, how many items total?',
935
1001
  });
936
1002
 
937
1003
  console.log('Answer:', text);
938
- // DeepSeek-R1 includes reasoning in the output with <think> tags:
939
- // <think>
940
- // First, I'll calculate the number of smaller boxes: 3 × 4 = 12
941
- // Then, the total items: 12 × 5 = 60
942
- // </think>
943
- // You have 60 items in total.
1004
+ console.log('Reasoning:', reasoningText);
944
1005
 
945
1006
  // Compare with reasoning disabled
946
- const modelNoReasoning = ollama('deepseek-r1:7b', { reasoning: false });
1007
+ const modelNoReasoning = ollama('deepseek-r1:7b', { think: false });
947
1008
  const { text: noReasoningText } = await generateText({
948
1009
  model: modelNoReasoning,
949
1010
  prompt: 'Calculate 3 × 4 × 5',
950
1011
  });
951
- // Output: 60 (without showing the thinking process)
1012
+ console.log('Answer:', noReasoningText);
952
1013
  ```
953
1014
 
1015
+ `think` controls the request, not filtering of the response. If the server returns thinking with `think` omitted or `false`, the adapter still preserves it as reasoning; it does not turn it into final text. If no thinking is returned, no reasoning content is added.
1016
+
1017
+ Ollama supplies a combined output token count. The adapter preserves that total and the raw counters, but leaves the text/reasoning token breakdown unknown instead of attributing all output tokens to text.
1018
+
954
1019
  **Recommended Reasoning Models**:
955
1020
 
956
1021
  - `deepseek-r1:7b` - Balanced performance and reasoning capability (5GB)
@@ -1 +1 @@
1
- {"version":3,"file":"generate-text.d.ts","sourceRoot":"","sources":["../../src/functions/generate-text.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,YAAY,IAAI,aAAa,EAAe,MAAM,IAAI,CAAC;AAqBhE;;GAEG;AACH,MAAM,WAAW,eAAe;IAC9B;;;OAGG;IACH,eAAe,CAAC,EAAE,OAAO,CAAC;IAE1B;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IAEzB;;;OAGG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAE9B;;;OAGG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAE3B;;;;;;;;;;;OAWG;IACH,+BAA+B,CAAC,EAAE,OAAO,CAAC;CAC3C;AAED;;GAEG;AACH,MAAM,MAAM,mBAAmB,GAAG,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CAAC,GAAG;IACtE;;OAEG;IACH,eAAe,CAAC,EAAE,eAAe,CAAC;CACnC,CAAC;AAEF;;;;;;;GAOG;AACH,wBAAsB,YAAY,CAChC,OAAO,EAAE,mBAAmB,GAC3B,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CA4NpD"}
1
+ {"version":3,"file":"generate-text.d.ts","sourceRoot":"","sources":["../../src/functions/generate-text.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,YAAY,IAAI,aAAa,EAAe,MAAM,IAAI,CAAC;AAsBhE;;GAEG;AACH,MAAM,WAAW,eAAe;IAC9B;;;OAGG;IACH,eAAe,CAAC,EAAE,OAAO,CAAC;IAE1B;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IAEzB;;;OAGG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAE9B;;;OAGG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAE3B;;;;;;;;;;;OAWG;IACH,+BAA+B,CAAC,EAAE,OAAO,CAAC;CAC3C;AAED;;GAEG;AACH,MAAM,MAAM,mBAAmB,GAAG,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CAAC,GAAG;IACtE;;OAEG;IACH,eAAe,CAAC,EAAE,eAAe,CAAC;CACnC,CAAC;AAEF;;;;;;;GAOG;AACH,wBAAsB,YAAY,CAChC,OAAO,EAAE,mBAAmB,GAC3B,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CAyNpD"}
@@ -1 +1 @@
1
- {"version":3,"file":"stream-text.d.ts","sourceRoot":"","sources":["../../src/functions/stream-text.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,IAAI,WAAW,EAAe,MAAM,IAAI,CAAC;AAE5D,KAAK,mBAAmB,GAAG,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAAC,CAAC;AAsC7D,MAAM,MAAM,iBAAiB,GAAG,mBAAmB,GAAG;IACpD,eAAe,CAAC,EAAE;QAChB;;WAEG;QACH,iBAAiB,CAAC,EAAE,OAAO,CAAC;QAC5B;;WAEG;QACH,wBAAwB,CAAC,EAAE,OAAO,CAAC;QACnC;;WAEG;QACH,eAAe,CAAC,EAAE,MAAM,CAAC;QACzB;;WAEG;QACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;KAC3B,CAAC;CACH,CAAC;AAUF,wBAAsB,UAAU,CAC9B,OAAO,EAAE,iBAAiB,GACzB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAsclD"}
1
+ {"version":3,"file":"stream-text.d.ts","sourceRoot":"","sources":["../../src/functions/stream-text.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,IAAI,WAAW,EAAe,MAAM,IAAI,CAAC;AAG5D,KAAK,mBAAmB,GAAG,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAAC,CAAC;AAsC7D,MAAM,MAAM,iBAAiB,GAAG,mBAAmB,GAAG;IACpD,eAAe,CAAC,EAAE;QAChB;;WAEG;QACH,iBAAiB,CAAC,EAAE,OAAO,CAAC;QAC5B;;WAEG;QACH,wBAAwB,CAAC,EAAE,OAAO,CAAC;QACnC;;WAEG;QACH,eAAe,CAAC,EAAE,MAAM,CAAC;QACzB;;WAEG;QACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;KAC3B,CAAC;CACH,CAAC;AAUF,wBAAsB,UAAU,CAC9B,OAAO,EAAE,iBAAiB,GACzB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAkclD"}