langfx.js 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/LICENSE +202 -0
  2. package/NOTICE +11 -0
  3. package/README.md +107 -0
  4. package/dist/agentic.d.ts +128 -0
  5. package/dist/agentic.js +265 -0
  6. package/dist/agentic.js.map +1 -0
  7. package/dist/cache.d.ts +32 -0
  8. package/dist/cache.js +135 -0
  9. package/dist/cache.js.map +1 -0
  10. package/dist/cancellation.d.ts +9 -0
  11. package/dist/cancellation.js +70 -0
  12. package/dist/cancellation.js.map +1 -0
  13. package/dist/errors.d.ts +40 -0
  14. package/dist/errors.js +44 -0
  15. package/dist/errors.js.map +1 -0
  16. package/dist/index.d.ts +18 -0
  17. package/dist/index.js +15 -0
  18. package/dist/index.js.map +1 -0
  19. package/dist/json-stream.d.ts +73 -0
  20. package/dist/json-stream.js +222 -0
  21. package/dist/json-stream.js.map +1 -0
  22. package/dist/langfunc.d.ts +20 -0
  23. package/dist/langfunc.js +28 -0
  24. package/dist/langfunc.js.map +1 -0
  25. package/dist/language-model.d.ts +78 -0
  26. package/dist/language-model.js +218 -0
  27. package/dist/language-model.js.map +1 -0
  28. package/dist/llms/anthropic.d.ts +30 -0
  29. package/dist/llms/anthropic.js +365 -0
  30. package/dist/llms/anthropic.js.map +1 -0
  31. package/dist/llms/gemini.d.ts +34 -0
  32. package/dist/llms/gemini.js +380 -0
  33. package/dist/llms/gemini.js.map +1 -0
  34. package/dist/llms/images.d.ts +10 -0
  35. package/dist/llms/images.js +43 -0
  36. package/dist/llms/images.js.map +1 -0
  37. package/dist/llms/index.d.ts +7 -0
  38. package/dist/llms/index.js +4 -0
  39. package/dist/llms/index.js.map +1 -0
  40. package/dist/llms/openai.d.ts +30 -0
  41. package/dist/llms/openai.js +398 -0
  42. package/dist/llms/openai.js.map +1 -0
  43. package/dist/llms/transport.d.ts +4 -0
  44. package/dist/llms/transport.js +90 -0
  45. package/dist/llms/transport.js.map +1 -0
  46. package/dist/mapping.d.ts +71 -0
  47. package/dist/mapping.js +190 -0
  48. package/dist/mapping.js.map +1 -0
  49. package/dist/message.d.ts +56 -0
  50. package/dist/message.js +86 -0
  51. package/dist/message.js.map +1 -0
  52. package/dist/python-preview.d.ts +24 -0
  53. package/dist/python-preview.js +368 -0
  54. package/dist/python-preview.js.map +1 -0
  55. package/dist/python-stream.d.ts +15 -0
  56. package/dist/python-stream.js +29 -0
  57. package/dist/python-stream.js.map +1 -0
  58. package/dist/python.d.ts +85 -0
  59. package/dist/python.js +728 -0
  60. package/dist/python.js.map +1 -0
  61. package/dist/query.d.ts +31 -0
  62. package/dist/query.js +151 -0
  63. package/dist/query.js.map +1 -0
  64. package/dist/retry.d.ts +12 -0
  65. package/dist/retry.js +56 -0
  66. package/dist/retry.js.map +1 -0
  67. package/dist/schema/zod.d.ts +4 -0
  68. package/dist/schema/zod.js +7 -0
  69. package/dist/schema/zod.js.map +1 -0
  70. package/dist/schema.d.ts +10 -0
  71. package/dist/schema.js +7 -0
  72. package/dist/schema.js.map +1 -0
  73. package/dist/template.d.ts +14 -0
  74. package/dist/template.js +75 -0
  75. package/dist/template.js.map +1 -0
  76. package/dist/testing/index.d.ts +33 -0
  77. package/dist/testing/index.js +56 -0
  78. package/dist/testing/index.js.map +1 -0
  79. package/dist/tool-call.d.ts +27 -0
  80. package/dist/tool-call.js +16 -0
  81. package/dist/tool-call.js.map +1 -0
  82. package/dist/tools.d.ts +43 -0
  83. package/dist/tools.js +155 -0
  84. package/dist/tools.js.map +1 -0
  85. package/docs/ANTHROPIC.md +43 -0
  86. package/docs/API_DESIGN.md +200 -0
  87. package/docs/GEMINI.md +68 -0
  88. package/docs/IMPLEMENTATION_STATUS.md +77 -0
  89. package/docs/LIVE_TESTING.md +24 -0
  90. package/docs/MAPPING.md +54 -0
  91. package/docs/OPENAI.md +40 -0
  92. package/docs/PORTING_PLAN.md +135 -0
  93. package/docs/PROMPT_PARITY.md +379 -0
  94. package/docs/PYTHON_PROTOCOL.md +109 -0
  95. package/docs/PYTHON_PROTOCOL_PARITY.md +964 -0
  96. package/docs/PYTHON_SCHEMA_EVALUATION.md +93 -0
  97. package/docs/PYTHON_STREAMING_PARITY.md +59 -0
  98. package/docs/RELEASING.md +25 -0
  99. package/docs/RETRIES_AND_CACHE.md +56 -0
  100. package/docs/SESSION_EVENTS.md +40 -0
  101. package/docs/SOURCE_AUDIT.md +133 -0
  102. package/docs/STREAMING.md +76 -0
  103. package/docs/TOOL_STREAMING.md +31 -0
  104. package/package.json +95 -0
@@ -0,0 +1,93 @@
1
+ # Python schema feature evaluation
2
+
3
+ Python commit: `5f1f0ebd556415e40a1a5e3e445654e309c265f2` (clean), Python 3.14.0. No inference, credentials, or generated-code execution tests.
4
+
5
+ ## Findings
6
+
7
+ Evaluated **39 feature categories** with **90 actual Python parsing probes**. The JS built-ins can represent 34 categories (some only by manually flattening/registering fields). Across 89 directly compared parsing probes, **4 differ in acceptance or value**. 33 schema representations match exactly. Unsupported features are not counted as passing comparisons. Of 56 jointly formatted accepted values, 1 differ in exact formatting.
8
+
9
+ The nine full-prompt fixtures still establish exact parity for those examples, not comprehensive schema parity. This wider evaluation exposes material gaps. A user-defined PythonCodec can implement more behavior, but is not counted as built-in support.
10
+
11
+ ## Feature matrix
12
+
13
+ | Feature | Python schema construction | JS representation | Exact schema text | Parsing agreement | Limit / difference |
14
+ | --- | --- | --- | --- | --- | --- |
15
+ | int | yes | available (see limit) | yes | 6/8 | Literal kinds and decimal syntax fixed; hex/underscore supported. Safe-integer bound and rejection of bool remain deliberate JS differences. |
16
+ | float | yes | available (see limit) | yes | 2/3 | JS rejects bool as a numeric value; Python permits it. |
17
+ | bool | yes | available (see limit) | yes | 3/3 | No differences in the selected probes; not exhaustive equivalence. |
18
+ | str | yes | available (see limit) | yes | 8/8 | Hex/Unicode, raw/triple/adjacent strings, octal and common escapes supported; full Python source grammar remains excluded. |
19
+ | none | yes | available (see limit) | yes | 1/1 | None renders NoneType; union annotations retain None. |
20
+ | optional | yes | available (see limit) | yes | 2/2 | Subclasses must be explicitly supplied with withSubclasses; no process-wide discovery. |
21
+ | union-raw | rejected | available (see limit) | no | not compared | JS explicit combinator works; raw Python top-level union is rejected in this checkout. |
22
+ | union-explicit | yes | available (see limit) | yes | 2/2 | Explicit subclass allowance renders inheritance metadata; no process-wide discovery. |
23
+ | union-number | yes | available (see limit) | yes | 2/2 | Fixed: union tries matching branches in order. |
24
+ | union-nested | yes | available (see limit) | yes | 2/2 | Flat nullable union tested; arbitrary flattening not claimed. |
25
+ | literal | yes | available (see limit) | yes | 2/2 | Built-in literal codec. |
26
+ | enum-spec | yes | available (see limit) | yes | 2/2 | Literal choices supported; Python enum class identity is separate. |
27
+ | enum-class | yes | unsupported | — | not compared | No automatic Python enum class introspection; use literal choices and an explicit factory. |
28
+ | any-raw | rejected | unsupported | — | not compared | Use explicit p.any; Python raw Any construction rejects. |
29
+ | any-spec | yes | available (see limit) | yes | 2/2 | Data literals only; unknown constructors are not permitted. |
30
+ | list | yes | available (see limit) | yes | 3/3 | No differences in the selected probes; not exhaustive equivalence. |
31
+ | list-shorthand | yes | available (see limit) | yes | 1/1 | Use list(int) rather than Python single-element schema-array shorthand. |
32
+ | list-length | yes | available (see limit) | yes | 3/3 | Built-in size constraints validate parse/format and export JSON bounds. |
33
+ | tuple-raw | rejected | unsupported | — | not compared | Use explicit tuple codec; raw Python tuple annotation rejects. |
34
+ | tuple-spec | yes | available (see limit) | yes | 3/3 | Tuple represented by a TS tuple array; normalized as tuple in the report. Python tuple/list syntax distinguished. |
35
+ | dict | yes | available (see limit) | yes | 2/3 | JS deliberately rejects duplicate keys; Python keeps the last. |
36
+ | dict-fixed | yes | available (see limit) | yes | 3/3 | Fixed-key record codec implemented. |
37
+ | dict-int-key | rejected | unsupported | — | not compared | String keys only in both schema paths. |
38
+ | set | yes | available (see limit) | yes | 7/7 | Primitive sets supported, including Python bool/numeric deduplication; JS iteration order can change formatting. Explicit item codecs enforce stricter validation than Python set[int]. |
39
+ | int-bounds | yes | available (see limit) | yes | 3/3 | Built-in integer bounds. |
40
+ | float-bounds | yes | available (see limit) | yes | 3/3 | Built-in numeric bounds. |
41
+ | str-regex | yes | available (see limit) | yes | 2/2 | Start-anchored portable patterns supported; JavaScript and Python regex dialects are not identical. |
42
+ | class | yes | available (see limit) | yes | 6/6 | Constructor/dictionary distinction fixed; subclass registration is explicit. |
43
+ | defaults | yes | available (see limit) | yes | 2/2 | Default values are validated and copied per instance. |
44
+ | descriptions | yes | available (see limit) | yes | 1/1 | Class/field descriptions supported. |
45
+ | inheritance | yes | available (see limit) | yes | 1/1 | Explicitly flattened fields; no automatic reflection of base classes. |
46
+ | recursive | yes | available (see limit) | yes | 1/1 | Named lazy references, bounded parsing, and JSON $defs implemented. |
47
+ | extra-fields | yes | available (see limit) | yes | 1/1 | Explicit extraFields schema validates additional keywords. |
48
+ | excluded-field | yes | available (see limit) | yes | 1/1 | Excluded prompt fields require defaults. |
49
+ | dataclass | yes | available (see limit) | yes | 1/1 | Explicit signature description and compact example format; no automatic dataclass introspection. |
50
+ | callable | rejected | unsupported | — | not compared | Callable execution remains out of scope; Python raw construction also rejects. |
51
+ | missing | yes | available (see limit) | yes | 2/2 | MISSING and Missing() reject in both response parsers; omitted fields still require defaults. |
52
+ | unknown | yes | available (see limit) | yes | 3/3 | UNKNOWN supported in Any; typed fields opt in with allowUnknown to preserve accurate TS return types. |
53
+ | assignment | yes | available (see limit) | yes | 2/2 | No differences in the selected probes; not exhaustive equivalence. |
54
+
55
+ ## Observed behavioral mismatches
56
+
57
+ | Feature | Response | Python | TypeScript |
58
+ | --- | --- | --- | --- |
59
+ | int | "True" | true | reject |
60
+ | int | "9007199254740993" | {"integer":"9007199254740993"} | reject |
61
+ | float | "True" | 1 | reject |
62
+ | dict | "{'a':1,'a':2}" | {"a":2} | reject |
63
+
64
+ ## Source-audited features beyond these probes
65
+
66
+ - **Descriptions and presentation:** Python class docstrings, Annotated field descriptions, constrained-type annotations, base classes, decorated methods (`include_method_in_prompt`), field exclusion (`exclude_from_prompt`), strict annotations, dependency allowlists, and natural-language formatting. JS now accepts explicit descriptions; method bodies and introspected annotations are not extracted.
67
+ - **Schema lifecycle:** Python annotation/ValueSpec conversion, dependency extraction (including recursion and non-PyGlove classes), defaults, extra-field policies, missing/unknown markers, protocol registration, schema/value rendering and validation. JS requires explicit schema/subtype registration and supports named lazy recursion, and explicit UNKNOWN opt-in, but has no automatic process-wide class registry or missing-value object model.
68
+ - **Output protocols:** Python supports JSON and Python v1/v2; code permissions, additional evaluation context, and bounded autofix exist in its Python parser. JS offers JSON and restricted Python data syntax, with no Python execution, repair, imports, additional execution context, or protocol-version selection. Those execution facilities should remain out of scope for browser core.
69
+ - **Streaming/provider conversion:** Python structured streaming and schema conversion into provider-native JSON constraints are separate integration surfaces. JS Python streaming emits raw text and best-effort frozen field previews with final validation; full incremental-parser parity remains unsupported; JSON/Zod and native adapters have their own supported subsets. This evaluation does not establish provider compatibility or model answer quality.
70
+
71
+ ## Additional evaluation evidence
72
+
73
+ The local Python `_base_test` and `_python_test` suites were run separately: 29 tests passed. They include fake correction/code-generation tests; the fixed-data differential harness itself performs no generated-code execution tests. No live inference was performed for this differential evaluation.
74
+
75
+ Python automatically includes known subclasses in dependency rendering. This harness declares Child(Foo), so Foo prompts include Child as well. The earlier nine parity captures did not register this subclass. That is a real context-dependent rendering difference, not a contradiction of the narrower fixture results.
76
+
77
+ ## Remaining priorities
78
+
79
+ 1. Correctness fixes are implemented for numeric unions, numeric literal kinds/syntax, and constructor-vs-dictionary parsing. Keep bool-as-number rejection, safe integers, and duplicate-key rejection explicit.
80
+ 2. Literal choices, descriptions, defaults, bounds, patterns, list sizes, and fixed-key dictionaries are implemented with validation. Add broader constraint combinations to future differential cases.
81
+ 3. Tuples, recursive schemas, common string syntax, and explicit subclass allowances are implemented. Primitive sets and explicit UNKNOWN are now implemented. Remaining gaps include automatic class introspection, automatic dataclass introspection and set ordering.
82
+ 4. Differential Python streaming fixtures now cover 24 traces; see PYTHON_STREAMING_PARITY.md. The seven-request live Gemini 3.7 Flash suite passed on 2026-09-16, including default-protocol queries and streaming; see LIVE_TESTING.md. Unfinished-string deltas, tuple/grouping, and primitive set previews are implemented. Keep arbitrary code evaluation/method execution outside the browser runtime.
83
+
84
+ ## Reproduce and scope
85
+
86
+ ```sh
87
+ ../langfx/.venv/bin/python scripts/capture-schema-evaluation.py ../langfx
88
+ npm run evaluate:schemas
89
+ ```
90
+
91
+ The Python fixture records actual construction/render/parse/format outcomes. The JS report compares built-in or explicitly flattened equivalents and writes detailed per-probe evidence to tests/fixtures/python-schema-evaluation-results.json. The evaluation command reports known gaps. `npm run check:schemas` enforces the captured feature/probe inventory, exact supported schema text, and parsing/example agreement with five individually documented exceptions (four deliberate rejections and set ordering). It runs in `npm run check`, never rewrites evidence, and fails on stale exceptions. Unsupported features remain explicit exclusions, not passing parity claims. Existing strict prompt-parity checks remain separate.
92
+
93
+ Source audit: `langfx/structured/schema/_base.py`, `_python.py`, `_base_test.py`, `_python_test.py`, and the local JS `src/python.ts`, `src/schema.ts`, `src/mapping.ts`. “Comprehensive” here means coverage of the Langfx schema surface found in these sources, not every possible Python type, value, or dependency-library behavior. Source-only entries above are not experimentally verified by this harness.
@@ -0,0 +1,59 @@
1
+ # Python streaming comparison
2
+
3
+ Source: Python commit `33e118df543ea45d338c0eaafa687420b92288a6` (clean), Python 3.14.0. Fixed fake-model responses through aquery_stream; no live inference.
4
+
5
+ 14/24 normalized semantic traces match across 12 responses, each streamed whole and one character at a time. This is not exact event/timing parity.
6
+
7
+ ## Explicit normalization
8
+
9
+ - Remove JS raw textDelta and container objectStart/objectEnd events: Python emits object events only for class instances.
10
+ - Compare Python instances with JS frozen constructor descriptors by class name and fields.
11
+ - Coalesce consecutive string deltas at the same path. Chunk-level emission timing is not compared.
12
+ - Omit Python value_spec, error details, usage and finish reason; these are not covered by this comparison.
13
+
14
+ | Response | Mode | Projected events | Python terminal | JS terminal |
15
+ | --- | --- | --- | --- | --- |
16
+ | class | whole | match | generationComplete | generationComplete |
17
+ | class | characters | match | generationComplete | generationComplete |
18
+ | escaped | whole | match | generationComplete | generationComplete |
19
+ | escaped | characters | match | generationComplete | generationComplete |
20
+ | nested | whole | match | generationComplete | generationComplete |
21
+ | nested | characters | match | generationComplete | generationComplete |
22
+ | empty | whole | match | generationComplete | generationComplete |
23
+ | empty | characters | match | generationComplete | generationComplete |
24
+ | union | whole | documented difference | parsingFailed | generationComplete |
25
+ | union | characters | documented difference | parsingFailed | generationComplete |
26
+ | fenced | whole | match | generationComplete | generationComplete |
27
+ | fenced | characters | match | generationComplete | generationComplete |
28
+ | raw | whole | documented difference | parsingFailed | generationComplete |
29
+ | raw | characters | documented difference | parsingFailed | generationComplete |
30
+ | triple | whole | match | generationComplete | generationComplete |
31
+ | triple | characters | match | generationComplete | generationComplete |
32
+ | adjacent | whole | documented difference | parsingFailed | generationComplete |
33
+ | adjacent | characters | documented difference | parsingFailed | generationComplete |
34
+ | wrong-type | whole | documented difference | parsingFailed | parsingFailed |
35
+ | wrong-type | characters | documented difference | parsingFailed | parsingFailed |
36
+ | incomplete | whole | match | parsingFailed | parsingFailed |
37
+ | incomplete | characters | match | parsingFailed | parsingFailed |
38
+ | trailing-code | whole | documented difference | generationComplete | parsingFailed |
39
+ | trailing-code | characters | documented difference | generationComplete | parsingFailed |
40
+
41
+ ## Differences
42
+
43
+ - **union:** Python streaming rejects the tested top-level union; JS resolves registered constructors.
44
+ - **raw:** Python streaming rejects raw string syntax; JS accepts it.
45
+ - **adjacent:** Python streaming rejects adjacent string literals; JS concatenates them.
46
+ - **trailing-code:** Python streaming ignores trailing text; JS rejects it at final validation.
47
+ - **wrong-type:** Both reject, but JS exposes unvalidated previews before final schema validation; Python validates during parsing.
48
+
49
+ Python capture reported 6 asynchronous cleanup diagnostics. These are recorded in fixture provenance; they are not counted as event mismatches.
50
+
51
+ ## Reproduce
52
+
53
+ ```sh
54
+ ../langfx/.venv/bin/python scripts/capture-python-streaming.py ../langfx
55
+ npm run check:streaming-parity
56
+ node scripts/check-python-streaming.mjs --report
57
+ ```
58
+
59
+ The check runs offline from captured evidence, rejects lost cases, enforces matching projected traces and documented terminal outcomes, and detects stale differences. --report explicitly updates the evidence report; ordinary checks never rewrite it. This fixture does not establish parity for all schema features, cancellation, provider metadata, or real-model behavior.
@@ -0,0 +1,25 @@
1
+ # Initial alpha: 0.1.0-alpha.0
2
+
3
+ Release candidate for npm package `langfx.js`, published under the `alpha` dist-tag. The tag intentionally leaves `latest` untouched. Publication has not yet occurred.
4
+
5
+ ## Included
6
+
7
+ Async TypeScript APIs for messages, templates, language functions, typed queries, structured streaming, tools, agents, and invocation traces. Python constructor syntax is the default structured-query protocol; JSON is explicit. Gemini, Anthropic, and OpenAI adapters support fixture-tested image inputs and local tool continuations. The package is ESM with TypeScript declarations, targets Node.js 22+ and browser/worker hosts, and has no required runtime dependencies. Zod is an optional peer.
8
+
9
+ ## Validation and limitations
10
+
11
+ Run `npm ci` and `npm run check`. Checks cover runtime tests, type contracts, browser/worker Web-API realms, Python prompt/schema/streaming comparisons, isolated package imports/declarations, optional Zod integration, and the packaged README quick-start.
12
+
13
+ Gemini's seven-request live suite passed on 2026-09-16. OpenAI/Anthropic live inference and real-provider browser CORS remain unverified. Mocked browser/worker demos do not establish provider CORS support. Apps own authentication and transport; do not embed shared provider secrets in publicly delivered code.
14
+
15
+ This is an alpha, with no stable API guarantee. Python schemas support a documented subset, and parsing does not evaluate generated code. Memory abstractions, persistence, MCP, embeddings, and React bindings are outside this release. See IMPLEMENTATION_STATUS.md and the Python parity reports for the precise scope.
16
+
17
+ ## Prepare and publish
18
+
19
+ 1. Confirm a clean release checkout, the intended version, npm account, and permission to publish the package. `npm whoami --registry=https://registry.npmjs.org/` verifies login; authenticate interactively with `npm login` when needed.
20
+ 2. Run `npm run check`. `npm pack --pack-destination <artifact-directory>` builds the declarations and JavaScript through `prepack`. Inspect the archive's file list and SHA-512 integrity. Only `dist`, docs, README, license/notice, and npm-required metadata belong in the archive.
21
+ 3. Review the exact archive with `npm publish <archive.tgz> --dry-run --tag alpha --access public --registry=https://registry.npmjs.org/`. Dry-run success does not establish publish rights or reserve the name.
22
+ 4. After publication approval, publish that reviewed archive with the same command without `--dry-run`. Directory publication additionally runs `prepublishOnly` to enforce `npm run check`; archive publication relies on the checks already completed for that archive. npm may require an interactive authentication challenge.
23
+ 5. Verify `npm view langfx.js@0.1.0-alpha.0 version dist.integrity --json` matches the archive, and `npm dist-tag ls langfx.js` maps `alpha` to the released version. Install the exact version in a clean consumer and smoke-test it before marking the release complete.
24
+
25
+ Do not move the `latest` tag as part of this alpha. Commit/tag the exact released source and update publication status after the registry verification succeeds. Never include npm credentials in the repository or archive.
@@ -0,0 +1,56 @@
1
+ # Retries and response caching
2
+
3
+ `LanguageModel` owns retry and cache behavior for every subclass. Both are opt-in. Configure defaults on the model or override them on `call`, `sample`, or `query`. I/O stays asynchronous even for cache hits.
4
+
5
+ ```ts
6
+ import * as lf from 'langfx.js';
7
+
8
+ const cache = new lf.InMemoryCache({ maxEntries: 500, maxChars: 8_388_608, ttl: 300 });
9
+ const lm = new lf.llms.Gemini({
10
+ model: 'your-model-id',
11
+ apiKey: async signal => obtainAppCredential(signal), // application-owned auth
12
+ cache,
13
+ cacheNamespace: 'current-user-and-workspace',
14
+ maxAttempts: 3,
15
+ retryInterval: [1, 2],
16
+ exponentialBackoff: true,
17
+ maxRetryInterval: 30,
18
+ });
19
+ const first = await lm.call('Explain closures.');
20
+ const second = await lm.call('Explain closures.');
21
+ console.log(second.metadata['cacheHit'], second.metadata['cachedUsage']);
22
+ await lm.call('Explain closures.', { cacheSeed: null }); // bypass cache
23
+ await cache.clear();
24
+ ```
25
+
26
+ ## Retry policy
27
+
28
+ The names and seconds-based intervals map to Python's `max_attempts`, `retry_interval`, `exponential_backoff`, and `max_retry_interval`. `maxAttempts` includes the initial request. Unlike Python's default five attempts, TypeScript defaults to **one attempt**, making retries explicit. The default interval is uniformly sampled from `[1, 2]` seconds, exponential backoff is enabled, and waits are capped at 30 seconds. A fixed interval is also supported. Per-call overrides inherit other model defaults.
29
+
30
+ Only `RateLimitError` and `TemporaryLMError` retry. Authentication failures, malformed output, schema validation failures, cancellation, generic errors, and unclassified fetch failures do not. Providers should classify transient failures explicitly. The last failure is preserved when attempts are exhausted. Failed attempt signals are aborted before backoff; cancellation interrupts both requests and waiting timers. An application Session deadline reaches these same signals.
31
+
32
+ With retries or caching enabled, `sample` handles prompts individually in input order. A later failure cannot cause an earlier completed prompt to be resubmitted. Without either feature, provider batching remains available. No Action or application tool is automatically retried.
33
+
34
+ Streams may retry only before their first yielded chunk. Even an empty chunk counts as visible output. Each attempt is closed before starting another. After any chunk, failures propagate through the existing stream contract. Stream caches and replay are not implemented; `stream` and `queryStream` bypass response caching.
35
+
36
+ Retry-After headers, provider-wide concurrency limits, retry trace events, and aggregate usage for failed attempts are not implemented. Successful results report the provider's successful-attempt usage; providers may bill failed attempts without reporting their usage, so this is not a complete billing ledger.
37
+
38
+ ## Cache contract and isolation
39
+
40
+ The base class is named `Cache`, matching Python. Its `get`, `put`, `delete`, and `clear` methods return Promises. This initial TypeScript storage contract receives canonical string keys and completed `SamplingResult` objects; Python's cache receives model/prompt/seed arguments. The base model constructs keys centrally. `InMemoryCache` is the flat TypeScript equivalent of Python's `llms.cache.InMemory`.
41
+
42
+ Cache keys include a version, model instance identity, model ID, application namespace, seed, message roles/parts/metadata, and merged output-affecting sampling settings (including tools and response schema). Object key order is canonicalized. Credentials, transport objects, cancellation signals, and retry settings are not included. Separate model instances never share entries, even if their model IDs match; this conservatively prevents cross-endpoint or cross-credential reuse. Change `cacheNamespace` or clear the cache if a single instance's credentials, routing, or application context changes. Keys are runtime-local and are not a persistent cross-process format.
43
+
44
+ `cacheSeed` defaults to zero. Different integer seeds select separate entries without changing provider sampling. `cacheSeed: null` or `cache: null` bypasses caching. Only complete, cardinality-checked model responses are retained. Query schemas validate again on every hit. Mapping/parsing/postprocessing failures evict the associated response entry, including for native structured queries. A later invocation makes a fresh model request. `lm.invalidateCache(response)` removes the entry associated with that model call; bookkeeping is private and does not expose cache keys in message metadata. Tool calls in a model response may be cached, but tool executions and Action results are never cached.
45
+
46
+ The memory implementation stores private serialized snapshots and returns fresh messages, tool calls, metadata, and byte arrays. Callers cannot poison later hits by mutating returned values. Plain data, standard messages, image URIs, and image bytes are supported. Blob inputs, custom message subclasses, functions, cycles, accessors in plain records, and unsupported metadata types bypass retention. URI media keys identify the URI, not remotely fetched content; use a new seed or bypass caching when that content changes. Treat request objects/options as immutable while a request is pending, and clear the cache if a custom model changes its internal output-affecting state.
47
+
48
+ The cache evicts least-recently-used entries at its entry or retained-character limit. TTL starts at insertion and is not extended by reads. Expired entries are removed lazily. Defaults are 1,000 entries, 8,388,608 UTF-16 code units for keys plus values, and 300 seconds. Oversized entries are skipped; zero TTL disables retention. Limits bound retained serialized data, not temporary serialization allocations. Storage is process/page memory only; nothing is persisted. Concurrent misses are independent requests, not coalesced, so cancelling one caller does not cancel another.
49
+
50
+ Custom cache implementations must return independent snapshots. Their storage errors propagate without retrying model/tool work. Pending cache operations are raced against cancellation; an external implementation is responsible for its own underlying cleanup and may finish an already-started write after cancellation.
51
+
52
+ ## Usage accounting
53
+
54
+ A hit has `cacheHit: true`, `usage` with zero newly consumed tokens, and optional `cachedUsage` containing the original response's provider usage. `LanguageModel.call` carries these fields into message metadata. No original token counts are invented when they were unavailable.
55
+
56
+ Session query traces retain `cacheHit` and `cachedUsage`. `session.usageSummary` counts newly reported usage; `session.cachedUsageSummary` separately totals original usage for cache hits and identifies hits whose original usage was unknown. Neither is a pricing calculation.
@@ -0,0 +1,40 @@
1
+ # Session logs, progress, and metadata
2
+
3
+ Actions use their invocation-bound `Session` to report status without a UI dependency:
4
+
5
+ ```ts
6
+ class Search extends lf.Action<string> {
7
+ async call(session: lf.Session): Promise<string> {
8
+ session.addMetadata({ operation: 'search' });
9
+ session.info('Searching', { metadata: { source: 'catalog' } });
10
+ session.updateProgress('Loading results', { completed: 0, total: 1 });
11
+ const result = await session.query('Suggest a search term.');
12
+ session.updateProgress('Finished', { completed: 1, total: 1 });
13
+ return result;
14
+ }
15
+ }
16
+
17
+ const session = new lf.Session({ lm, maxLogEntries: 1000 });
18
+ const unsubscribe = session.subscribe(event => {
19
+ if (event.type === 'progress') renderStatus(event.invocation.id, event.progress.title);
20
+ if (event.type === 'log') renderLog(event.entry);
21
+ });
22
+ try { await new Search().invoke(session); }
23
+ finally { unsubscribe(); await session.dispose(); }
24
+ ```
25
+
26
+ `debug`, `info`, `warning`, `error`, and `fatal` preserve Python Session's logging levels. These methods only report logs: even `fatal` does not throw or terminate an action. Options replace Python keyword arguments: `{ keep: false, metadata: { ... } }` emits a log event without retaining it. No console output is automatic. `updateProgress(title, metadata)` ports Python `update_progress`; metadata is application-defined rather than an enforced percentage schema.
27
+
28
+ Each trace node holds its retained `logs` and latest `progress`. `session.allLogs` returns retained entries in emission order across the session, including concurrent branches. An entry's `invocationId` identifies its owning invocation, while `sequence` orders events even when timestamps coincide. Progress emits every update but retains only the latest value per node.
29
+
30
+ `SessionEvent` discriminates `start`, `end`, `log`, `progress`, and `metadata`. Events carry frozen trace snapshots; log/progress metadata is shallow-copied and frozen. Nested application values remain application-owned. Subscribers run synchronously in registration order; returned promises are observed for failures but not awaited. Errors are isolated in `observerErrors`, and subscriptions do not provide backpressure.
31
+
32
+ Retained logs are bounded by `maxLogEntries` (default 10,000 per session; zero disables retention). Exceeding the bound throws before emitting or storing a retained entry. Use `keep: false` for transient events at that point. This bounds entry count, not message bytes or arbitrary metadata. Ended invocation views, disposed sessions, and cancelled runs reject further log/progress updates. Applications own retention and rendering outside the session.
33
+
34
+ ## Invocation metadata
35
+
36
+ `session.addMetadata({ key: value })` shallow-merges annotations into the current invocation, matching Python `Session.add_metadata`. A repeated key replaces its previous value; nested records are not recursively merged. Each update emits a `metadata` event whose `invocation.metadata` contains the new snapshot.
37
+
38
+ `session.metadata` always reads root session metadata, matching Python's getter. `session.currentInvocation` reads the snapshot owned by the current view, so actions use `session.currentInvocation.metadata` to inspect their own annotations. Calls on the owning Session annotate its root. Child actions and concurrent branches begin with empty metadata; annotations are neither inherited nor copied to parents.
39
+
40
+ Metadata records are shallow-copied and frozen. Earlier records and event snapshots retain their previous top-level values; nested application values are shared. Metadata is not persisted, byte-limited, or guaranteed JSON-serializable. Applications should store small annotations and handle serialization explicitly. Ended or cancelled views reject updates; snapshots remain readable after completion.
@@ -0,0 +1,133 @@
1
+ # Source audit
2
+
3
+ Examined September 16, 2026. Local checkout: `/Users/daiyip/git/langfx`, remote `https://github.com/free-solo/langfx.git`. Baseline: `2a1ea4e8dcd7075bf745ad707f901ece8546f47d` (September 12, 2026, “Capture learnings from #159 (#160)”). The checkout was clean during inspection and was not modified.
4
+
5
+ This audit inventories all Python modules and examines the public exports, dependencies, key runtime implementations, and selected regression tests. It is not a claim that every implementation line was reviewed or that the Python test suite was run. Links below are pinned to the examined commit.
6
+
7
+ ## Inventory
8
+
9
+ Counts cover `langfx/**/*.py`; tests are files ending in `_test.py`. Other files include package initializers and helper modules. Non-test line counts include comments and documentation, so they measure source surface rather than porting effort. External integration tests in `testing/` and notebook tutorials are outside these counts.
10
+
11
+ | Area | Other Python files | Test files | Non-test lines |
12
+ | --- | ---: | ---: | ---: |
13
+ | agentic | 6 | 2 | 3,214 |
14
+ | coding | 7 | 5 | 1,050 |
15
+ | core | 18 | 16 | 9,447 |
16
+ | data | 5 | 3 | 898 |
17
+ | ems | 4 | 3 | 729 |
18
+ | eval | 24 | 16 | 6,981 |
19
+ | llms | 27 | 17 | 9,328 |
20
+ | mcp | 4 | 3 | 804 |
21
+ | memories | 2 | 1 | 97 |
22
+ | modalities | 6 | 5 | 773 |
23
+ | structured | 21 | 15 | 8,744 |
24
+ | **Total** | **124** | **86** | **42,065** |
25
+
26
+ ## Findings that determine the port
27
+
28
+ ### The README understates the current runtime
29
+
30
+ The introductory material describes PyGlove/Jinja/requests, while current requirements use pygx, anyio, httpx, MCP, MIME detection, and cloudpickle. The port scope must follow the code and requirements rather than just the introductory example. Source: [README.md](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/README.md).
31
+
32
+ ### The public API is much larger than query
33
+
34
+ Exports include messages, language functions, model/cache/usage interfaces, embeddings, structured helpers, code generation, actions, sessions, modalities, concurrency wrappers, and evaluation. Re-exporting all of it would make a browser package unnecessarily broad. Source: [langfx/__init__.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/__init__.py).
35
+
36
+ ### PyGlove is architectural, not a removable type annotation
37
+
38
+ Component behavior includes symbolic traversal/replacement and contextual settings. Other modules depend on pg.Object schemas, references, lifecycle hooks, topology, and HTML rendering. Preserve the useful contracts with explicit data/configuration rather than trying to shim pg.Object. Source: [langfx/_component.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_component.py).
39
+
40
+ ### Messages preserve role and modality structure
41
+
42
+ Message chunks support nested role messages, markup, normalization, provenance, conversion registries, and a content identity. Flattening everything to text would lose system/user distinctions, attachments, and cache correctness. Source: [langfx/_message.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_message.py).
43
+
44
+ ### Templates carry structured objects through rendering
45
+
46
+ Jinja composition, partial rendering, contextual variables, and modalities are integrated. Retain Template and basic interpolation/composition. Full Jinja remains outside the proposed scope; prompt functions or optional tagged templates can handle more elaborate composition. Source: [langfx/_template.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_template.py).
47
+
48
+ ### Language functions are reusable prompt pipelines
49
+
50
+ LangFunc adds model invocation, input/output transforms, async calls, and streaming to Template. Retain its class and transform hooks; move last-input/output state to invocation records in TypeScript. Source: [langfx/_langfunc.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_langfunc.py).
51
+
52
+ ### Two structured-generation paths exist
53
+
54
+ aquery supports prompt-engineered protocols and an opt-in native structured-output path, plus examples, system sections, multiple models/samples, autofix, and tracing. Default protocol behavior is Python-oriented; JSON-first TypeScript is an explicit divergence. Source: [langfx/structured/_querying.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/_querying.py).
55
+
56
+ ### Streaming already has a useful UI contract
57
+
58
+ ObjectStart, FieldStart, StringFieldDelta, FieldEnd, ObjectEnd, GenerationComplete, and ParsingFailed are valuable semantics. Replace pg.KeyPath, pg.ValueSpec, and Python class objects with portable paths/schema references and plain data. Source: [langfx/structured/streaming/_events.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/streaming/_events.py).
59
+
60
+ ### Provider schemas require transformations
61
+
62
+ Tool and response schemas share conversion. The source handles type discriminators, root unions, const/enum normalization, and strict required fields. TypeScript needs explicit adapter compatibility rules, not a cast from arbitrary JSON Schema into every provider payload. Source: [langfx/_tool_schema.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_tool_schema.py).
63
+
64
+ ### Tool declarations are not an autonomous loop
65
+
66
+ SamplingOptions declares tools and tool choice. ToolCall contains ID/name/raw args and as_tool constructs a matching typed object. Consumers still execute and feed back results. The proposed bounded loop is new orchestration over these primitives. Source: [langfx/_language_model.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_language_model.py).
67
+
68
+ ### Actions are async; sessions still contain Python bridging
69
+
70
+ Action.call is async and invocation records capture work/results/errors. Session.aquery currently wraps the synchronous structured query through a worker with context propagation. TypeScript should directly await its native async query path. Source: [langfx/agentic/_action.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/agentic/_action.py).
71
+
72
+ ### Trace data and rendering are intertwined
73
+
74
+ ActionInvocation, ExecutionTrace, and ParallelExecutions contain useful hierarchy/usage/timing concepts alongside PyGlove HTML controls. Port the data/events; implement UI separately. Source: [langfx/agentic/_tracing.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/agentic/_tracing.py).
75
+
76
+ ### Conversation memory is deliberately small
77
+
78
+ The built-in memory stores user/assistant pairs with max_turns and text recollection. It is not a vector store or durable agent state. Memory abstractions are excluded from the current TypeScript scope; applications own their conversation history. Source: [langfx/memories/_conversation_history.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/memories/_conversation_history.py).
79
+
80
+ ### Cache identity is more than prompt text
81
+
82
+ default_key includes content_key, effective sampling options, and seed; the in-memory implementation also partitions by model ID. Preserve these semantics and add explicit provider/endpoint/application namespaces where needed. Source: [langfx/llms/cache/_base.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/cache/_base.py).
83
+
84
+ ### MCP spans three different environments
85
+
86
+ The client offers command/stdio, URL/HTTP, and in-process FastMCP construction. Only remote HTTP belongs directly in a browser port. The tool wrapper generates runtime Python classes and converts structured/multimodal tool results. Source: [langfx/mcp/_client.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/mcp/_client.py).
87
+
88
+ ### Evaluation is a separate subsystem
89
+
90
+ Experiments, metrics, example state, reports, checkpointing, sequential/parallel/debug/Beam runners, and symbolic expansion are a large independent surface. A small async metric runner is a better later app-oriented subset. Source: [langfx/eval/_evaluation.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/eval/_evaluation.py).
91
+
92
+ ### Code generation includes execution machinery
93
+
94
+ function_gen generates Python implementations and validates them with tests; coding/python adds parsing, execution, correction, and multiprocessing sandboxing. Translating that to JavaScript execution is a new isolation design, not a necessary language-function feature. Source: [langfx/structured/_function_generation.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/_function_generation.py).
95
+
96
+ ### On-device support is platform-specific
97
+
98
+ Apple Intelligence uses a platform SDK. The requirements mark it as an opt-in macOS-specific extra. This does not become browser inference through TypeScript translation. Source: [langfx/llms/_apple_intelligence.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/_apple_intelligence.py).
99
+
100
+ ### llama.cpp support is remote
101
+
102
+ LlamaCppRemote talks to a serving endpoint. An in-browser WASM/WebGPU runtime would be a new adapter and deployment capability. Source: [langfx/llms/_llama_cpp.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/_llama_cpp.py).
103
+
104
+ Additional supporting files: [requirements.txt](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/requirements.txt), [langfx/agentic/_session.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/agentic/_session.py), [langfx/agentic/_event_handler.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/agentic/_event_handler.py), [langfx/structured/schema/_json.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/schema/_json.py), [langfx/llms/_rest.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/_rest.py), [langfx/llms/cache/_in_memory.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/cache/_in_memory.py), [langfx/mcp/_tool.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/mcp/_tool.py), [langfx/_embedding_model.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_embedding_model.py), [langfx/eval/runners/_beam.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/eval/runners/_beam.py), [langfx/coding/python/_sandboxing.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/coding/python/_sandboxing.py).
105
+
106
+ ## Regression behavior to carry forward
107
+
108
+ Use these tests as sources for language-neutral fixtures and contract assertions. Port their intent; do not copy Python runtime or HTML-specific assertions.
109
+
110
+ | Source test area | Behavior to preserve |
111
+ | --- | --- |
112
+ | [langfx/_message_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/_message_test.py) | Nested roles, system sections, literal braces, modality identity, and content ordering survive normalization. |
113
+ | [langfx/structured/_querying_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/_querying_test.py) | Native output with root unions/lists/primitives; system-section hoisting; literal braces and markup preservation; parse failure closes the upstream generator (`test_parse_error_cancels_upstream_stream`). |
114
+ | [langfx/structured/streaming/_json_parser_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/structured/streaming/_json_parser_test.py) | Escapes and Unicode split across feeds, surrogate pairs, invalid escapes, polymorphic schemas, and terminal validation. |
115
+ | [langfx/llms/_rest_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/_rest_test.py) | SSE framing/comments/trailing events, HTTP error classification, and consumer-break cancellation (`test_consumer_break_propagates_generator_exit`). |
116
+ | [langfx/agentic/_action_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/agentic/_action_test.py) | Concurrent child overlap and correct trace ancestry, per-run state, deadlines, bounded concurrency, observer failure, and concurrent reuse (`test_same_instance_invoked_concurrently`). |
117
+ | [langfx/data/conversion/_gemini_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/data/conversion/_gemini_test.py) | Function calls/results, multimodal blocks, thought metadata, and unsupported content handling. Corresponding OpenAI/Anthropic converter tests cover their own wire semantics. |
118
+ | [langfx/llms/cache/_in_memory_test.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/langfx/llms/cache/_in_memory_test.py) | Cache partitioning, lookup, invalidation, and persistence behavior; replace filesystem persistence with a later storage adapter. |
119
+ | [testing/llms/test_streaming_tool_calls.py](https://github.com/free-solo/langfx/blob/2a1ea4e8dcd7075bf745ad707f901ece8546f47d/testing/llms/test_streaming_tool_calls.py) | Use as an integration-test reference for streamed tool calls; rewrite around browser transport and available credentials. |
120
+
121
+ ## Source documentation caveats
122
+
123
+ The Action docstring cautions against concurrent invocation of one instance, but the source now has a regression test explicitly checking that concurrent reuse does not clobber session ownership. Per-invocation state is therefore a deliberate TypeScript contract, without copying ambiguous last-run accessors.
124
+
125
+ Some surrounding Session examples still show synchronous calls, while Action.call is async. Session.aquery also retains a worker bridge even though structured.aquery exists. Follow implementations/tests when deciding what to preserve, and use one async TypeScript API.
126
+
127
+ The Python REST streaming base builds an aggregated text buffer; do not assume this alone proves support for every provider's interleaved candidates or tool deltas. The initial TypeScript request contract deliberately limits candidates to one; each advertised provider tool-streaming capability needs its own fixture suite.
128
+
129
+ ## Audit outcome
130
+
131
+ The strongest portable core is messages → composable prompts → validated queries/streams → traceable actions and tools. Most cross-language difficulty comes from runtime schemas, implicit context, provider conversion, cancellation, and application/tool lifecycle. Python execution, symbolic experimentation, and notebook rendering add substantial surface without helping that first client-side agent.
132
+
133
+ The feature matrix in [PORTING_PLAN.md](PORTING_PLAN.md) assigns a decision to each subsystem. The API in [API_DESIGN.md](API_DESIGN.md) makes the principal semantic changes explicit. This audit preceded implementation. The TypeScript core now has manually transcribed semantic fixtures tied to this source commit; Python is not a build or runtime dependency. See [implementation status](IMPLEMENTATION_STATUS.md).
@@ -0,0 +1,76 @@
1
+ # Structured streaming
2
+
3
+ `queryStream(prompt, schema, options)` returns an `AsyncIterable<StructuredStreamEvent<T>>` directly. It uses the same runtime schema, prompt rendering, and native-output selection as `query`. There are no synchronous or async-prefixed alternatives. The two-argument text overload continues to return `StreamChunk` values and throw on failure.
4
+
5
+ ```ts
6
+ import * as lf from 'langfx.js';
7
+ import { StaticResponse } from 'langfx.js/testing';
8
+ import { fromZod } from 'langfx.js/schema/zod';
9
+ import { z } from 'zod';
10
+
11
+ const Answer = fromZod(z.object({ explanation: z.string(), answer: z.number() }));
12
+ const lm = new StaticResponse('{"explanation":"Six times seven","answer":42}');
13
+
14
+ for await (const event of lf.queryStream('What is 6 × 7?', Answer, { lm, protocol: 'json' })) {
15
+ if (event.type === 'stringFieldDelta') {
16
+ console.log(event.path, event.delta); // display-only, unvalidated text
17
+ } else if (event.type === 'generationComplete') {
18
+ console.log(event.result.answer); // validated number
19
+ } else if ('error' in event) {
20
+ console.error(event.type, event.error);
21
+ }
22
+ }
23
+ ```
24
+
25
+ Replace the fake model with a streaming `LanguageModel` subclass. Set `nativeStructuredOutput: true` to send its JSON Schema through the provider's native constraints; otherwise select `protocol: 'json'` to include JSON generation instructions. Python-style output is the default for structured queries; it streams raw text and validates the complete response at the end. Python additionally provides best-effort field previews for constructors, lists, dictionaries, tuples, and primitive sets. Local validation always runs. Structured streams do not accept `returnsMessage`.
26
+
27
+ | Event `type` | Payload and meaning |
28
+ | --- | --- |
29
+ | `textDelta` | `delta`: raw Python protocol text, unvalidated and display-only |
30
+ | `fieldStart` | `path`: a field or array element begins (not emitted for the root) |
31
+ | `objectStart` | `path`, `kind`: an object or array begins |
32
+ | `stringFieldDelta` | `path`, `delta`: decoded string content, excluding object keys |
33
+ | `fieldEnd` | `path`, `value: unknown`: syntactically complete, unvalidated value |
34
+ | `objectEnd` | `path`, `kind`: `value: unknown`: object or array closes, before its `fieldEnd` (no root `fieldEnd`) |
35
+ | `generationComplete` | `result: T`, optional `usage` and `finishReason`: complete output passed synchronous schema validation |
36
+ | `parsingFailed` | `error: OutputValidationError`: invalid/incomplete output, limit exceeded, or schema validation failure; original cause retained |
37
+ | `generationFailed` | `error: unknown`: provider, transport, or cleanup error; provider error classification retained |
38
+ | `cancelled` | `error: CancelledError`: request cancellation |
39
+
40
+ Paths are immutable arrays of string keys and numeric array indices; `[]` is the root. Dots in keys remain literal. Both containers and primitives are supported at the root. String escapes can cross chunks; surrogate pairs are kept together in deltas. Lone surrogates follow JSON string semantics. Event boundaries need not match provider chunks. Unlike the Python events, these events omit runtime class/schema objects, accumulated `valueSoFar`, and buffered raw output. String deltas may be finer-grained than provider chunks; arrays also emit container events. There is no growing partial-object snapshot; apps can accumulate field events as needed.
41
+
42
+ Only `generationComplete` authorizes use as `T`. Schema transforms run once, on the complete JSON value, after the provider's final chunk. `fieldEnd` object/array values are recursively frozen so consumers cannot mutate the parser's validation input; custom schema parsers should construct their outputs without mutating input. Partial values may still violate the schema or be followed by a provider failure.
43
+
44
+ Strict JSON is required: no Markdown fences, comments, trailing commas, duplicate object keys, non-finite numbers, or trailing content. The duplicate-key and finite-number checks are intentionally stricter than `JSON.parse`. `maxChars` defaults to 1,048,576 UTF-16 code units, including whitespace; `maxDepth` defaults to 64 nested containers. Both options must be positive safe integers. These limits bound the parser, not the provider's own stream aggregation or network buffers.
45
+
46
+ Consume the stream through a terminal event to determine the outcome. A completed iteration delivers exactly one terminal outcome. Setup errors may throw before iteration starts. Early `break` or iterator `return()` closes the provider without delivering a terminal event. Parser errors and terminal outcomes close the provider before the terminal event is yielded. Pass `signal` to cancel pending I/O; custom providers must implement cooperative cancellation and cleanup. No retries or output repair occur.
47
+
48
+ Session-level streaming traces, streamed tool arguments, live-provider validation, and framework-specific UI bindings remain future work. Gemini integration is covered using mocked SSE responses.
49
+
50
+ ## Python protocol
51
+
52
+ ```ts
53
+ const schema = lf.python.record({ answer: lf.python.int });
54
+ for await (const event of lf.queryStream('What is 6 × 7?', schema, { lm })) {
55
+ if (event.type === 'textDelta') console.log(event.delta);
56
+ if (event.type === 'generationComplete') console.log(event.result.answer);
57
+ }
58
+ ```
59
+
60
+ Python mode emits raw `textDelta` progress, best-effort field previews, and a terminal event. Raw text deltas may split tokens, escapes, constructors, or fences; they are not parsed field values. Syntax and schema validation run once after the model's final chunk, so class factories never see incomplete syntax. Provider failures, cancellation, and early consumer exit do not invoke factories. Complete but schema-invalid output can invoke inner factories before outer validation fails, as with non-streaming parsing.
61
+
62
+ The accumulated response is bounded by `maxChars` (default and maximum 1,048,576 UTF-16 code units). `maxDepth` defaults to 64 and can be lowered, but cannot exceed 64 in Python mode. Character limits are checked as chunks arrive; syntax and nesting limits are checked at completion. `python.parse` accepts the same optional limits. Native structured output remains JSON-only. This is not full parity with Python's incremental object/field parser.
63
+
64
+ ### Python field previews
65
+
66
+ Constructor, list, tuple, primitive set, and string-keyed dictionary syntax can emit `objectStart`, `fieldStart`, `stringFieldDelta`, `fieldEnd`, and `objectEnd` before generation ends. Paths use field names and list indices. Constructor `objectStart` events include `pythonClass`; completed constructor previews use frozen `{ pythonClass, fields }` descriptors rather than application class instances. All nested preview data and paths are frozen. These values are syntactic, unvalidated display data; only `generationComplete.result` has the schema's result type.
67
+
68
+ String deltas emit while a quoted value is still arriving, including ordinary, raw, triple-quoted, and adjacent string literals. Incomplete escapes and ambiguous triple-quote endings are held until enough text arrives; UTF-16 high surrogates are held for a possible low surrogate, or flushed when the string closes or a following character resolves them. Concatenating the deltas yields the decoded string; delta boundaries depend on model chunks. Empty strings emit no string delta. Ambiguous tuple/grouping and set/dictionary first elements can still buffer their string events until a separator determines their path. Completed scalar fields wait for a delimiter. Tuple/grouping and primitive set previews are supported. Unsupported or malformed preview syntax stops further field previews, while raw text continues and the authoritative final parser still validates the response. Consequently, progress events need not form a balanced tree: a final error or fallback can leave a started preview unfinished. Treat the terminal event as authoritative.
69
+
70
+ Tuple and set `objectStart`/`objectEnd` events use `kind: 'array'` with `pythonKind: 'tuple' | 'set'`. Tuple preview values are frozen arrays; set preview values are frozen `{ pythonSet: readonly unknown[] }` descriptors, so consumers cannot mutate a live `Set`. Set element paths refer to textual positions, including duplicates; the completed descriptor deduplicates values (including `True`/`1` and `False`/`0`) in encounter order. Final results still use the registered schema's tuple or `Set` representation.
71
+
72
+ Ambiguous syntax delays some previews until a separator resolves it: `(value)` is grouping, `(value,)` is a tuple, and `{value}` is a set while `{'key': value}` is a dictionary. Events for an ambiguous first element are buffered until that distinction is known. Empty tuples `()`, empty sets `set()`, empty dictionaries `{}`, and trailing commas are supported.
73
+
74
+ ## Differential Python evidence
75
+
76
+ [The Python streaming comparison](PYTHON_STREAMING_PARITY.md) captures 12 fixed responses in whole-response and character chunks. `npm run check:streaming-parity` checks the normalized event traces and documented differences offline, and is included in `npm run check`. This compares semantic event order and values after explicit adaptations; it does not establish identical chunk timing or typed partial-object behavior.
@@ -0,0 +1,31 @@
1
+ # Tool-call streaming
2
+
3
+ `LanguageModel.stream()` and the text overload of `queryStream()` retain their existing `StreamChunk` API. A chunk may additionally carry a discriminated `toolEvent`:
4
+
5
+ - `toolCallStart`: provider call `id` and tool `name`.
6
+ - `toolCallDelta`: call `id` and raw JSON text `delta`. This text may be incomplete.
7
+ - `toolCallEnd`: a complete `ToolCall` with parsed object arguments. These have not been validated against a registered tool schema.
8
+
9
+ Tool events have empty text deltas. They are for display, not execution. Wait until the iterable completes with a successful `isFinal` chunk, then use `aggregated.toolCalls`, check registered names/policy/budgets, and validate with `Tool.prepare()` before executing. A tool-end event may be followed by truncation, cancellation, or provider failure. No tool executes automatically while streaming. The bounded Agent loop still uses complete, nonstreaming model calls.
10
+
11
+ Anthropic is the first supported adapter. It accumulates `input_json_delta` fragments, parses each closed block with the strict JSON parser, and retains the ordered completed response blocks in final continuation metadata. This follows the [Anthropic stream protocol](https://platform.claude.com/docs/en/build-with-claude/streaming). Duplicate IDs, invalid block ordering, non-object arguments, duplicate JSON keys, nonfinite numbers, and incomplete JSON reject. Blocks must be sequential; interleaved blocks reject under this adapter's protocol. Multiple tool calls in one message are supported.
12
+
13
+ Limits are 1,024 blocks, 1,048,576 UTF-16 units per tool's arguments, nesting depth 64, and 16,777,216 total content/argument/identifier units. SSE framing has its separate event limit. Empty argument streams represent `{}`. Completion is not schema validation. Gemini supports complete-call streaming as described below.
14
+
15
+ Final tool calls remain on `aggregated.toolCalls`, consistent with Python's final stream aggregation. The optional event union is a TypeScript addition for app displays. Intermediate aggregated messages contain text only. Stream failures throw; cleanup closes the reader, and retries stop after the first visible chunk, including a tool start. The schema overload of `queryStream()` is for structured response text, not tool events.
16
+
17
+ See [the browser/worker example](../examples/anthropic.ts) for a mocked stream → validate → execute → correlated tool result → answer round trip. It accepts an event callback for UI display. This checks wire fixtures, not live provider behavior.
18
+
19
+ ## OpenAI Responses
20
+
21
+ OpenAI emits the same events, correlating `output_index` and item `id` internally while exposing `call_id` for tool-result correlation. Calls may interleave. Argument-done events must match accumulated deltas, output-item completion must match the call identity and arguments, and the successful final response must match every streamed call at its original output index. Missing completion events or changed final arguments reject. This follows the [OpenAI function-calling stream protocol](https://developers.openai.com/api/docs/guides/function-calling).
22
+
23
+ Arguments use the same strict parser, 1,048,576-unit cap, and depth-64 limit. The adapter bounds output items at 1,024 and accumulated text/argument/identifier units at 16,777,216. Final aggregated messages retain encrypted reasoning and function items for scoped continuation. Partial aggregates contain no executable calls. See [the OpenAI example](../examples/openai.ts) for a mocked streamed round trip. Live behavior remains unverified.
24
+
25
+ ## Gemini generateContent
26
+
27
+ The existing `streamGenerateContent` adapter receives complete `functionCall` objects. It emits `toolCallStart` then `toolCallEnd` for each call, with no fabricated argument-delta events. Google documents the complete-chunk behavior in its [Interactions migration guide](https://ai.google.dev/gemini-api/docs/migrate-to-interactions). Partial-argument dialects and the Interactions API are outside this adapter's scope and are not silently treated as complete calls.
28
+
29
+ Calls without provider IDs receive local correlation IDs; these are not inserted into the original continuation. Wire IDs, original part order, and opaque thought signatures are retained, and function results use wire IDs only when supplied. Repeated idless function names represent separate calls. Duplicate IDs across chunks reject. Intermediate aggregates have no executable calls; final calls are exposed only after STOP and a clean end of stream. Trailing usage-only chunks are supported. The stream is bounded at 1,024 parts and 16 MiB of serialized parts. Arguments still require registered-tool schema validation.
30
+
31
+ See [the Gemini example](../examples/gemini-stream.ts) for the mocked streamed call → validated local tool → answer round trip used by the browser/worker demo. The Node live smoke suite also passed the streamed tool round trip with `gemini-3.7-flash` on 2026-09-16; see [live testing](LIVE_TESTING.md).