@nodaro/shared 2.19.0 → 2.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/index.cjs +548 -31
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.d.cts +1043 -386
  4. package/dist/index.d.ts +1043 -386
  5. package/dist/index.js +529 -32
  6. package/dist/index.js.map +1 -1
  7. package/package.json +1 -1
  8. package/src/__tests__/credit-identifiers.test.ts +133 -0
  9. package/src/__tests__/image-pricing-catalog-coverage.test.ts +139 -0
  10. package/src/__tests__/normalize-node-params.test.ts +36 -0
  11. package/src/__tests__/organizations-types.test.ts +29 -1
  12. package/src/__tests__/prompt-length-limits.test.ts +36 -0
  13. package/src/__tests__/safety-retry-policy.test.ts +47 -0
  14. package/src/__tests__/suno-credit-type.test.ts +54 -0
  15. package/src/__tests__/topaz-upscale.test.ts +132 -0
  16. package/src/__tests__/unresolved-ref-tokens.test.ts +66 -0
  17. package/src/__tests__/video-analysis-brief.test.ts +61 -0
  18. package/src/__tests__/video-analysis.test.ts +83 -0
  19. package/src/__tests__/video-audio-capability.test.ts +38 -4
  20. package/src/__tests__/video-catalog-totality.test.ts +85 -0
  21. package/src/__tests__/video-collapse-parity.test.ts +76 -0
  22. package/src/__tests__/video-ref-video-duration-limits.test.ts +69 -0
  23. package/src/__tests__/video-request-normalize.test.ts +239 -0
  24. package/src/credit-identifiers.ts +264 -20
  25. package/src/index.ts +30 -2
  26. package/src/model-catalog.ts +287 -6
  27. package/src/model-constants.ts +185 -13
  28. package/src/node-refs.ts +82 -0
  29. package/src/node-runtime-keys.ts +5 -0
  30. package/src/normalize-node-params.ts +8 -0
  31. package/src/organizations/types.ts +12 -0
  32. package/src/organizations/views.ts +114 -0
  33. package/src/safety-retry-policy.ts +37 -0
  34. package/src/topaz-upscale.ts +163 -0
  35. package/src/video-analysis.ts +115 -5
@@ -133,6 +133,24 @@ export interface ModelCatalogEntry {
133
133
  features?: readonly string[]
134
134
  aspectRatios?: readonly string[]
135
135
  resolutions?: readonly string[]
136
+ /**
137
+ * What this model actually RENDERS for a resolution outside `resolutions`.
138
+ *
139
+ * Most providers honour whatever band they are sent, so an off-list value is
140
+ * best snapped to the NEAREST supported one — a 4k request on a 1080p-max
141
+ * model renders 1080p. A few instead COLLAPSE anything unrecognised to a
142
+ * fixed default (MiniMax H3 → 2K, Wan 3.0 → 720p), and for those the nearest
143
+ * band is the wrong answer in the most expensive direction: snapping a stale
144
+ * "720p" to H3's cheap 768P would send a value the provider ignores, so we
145
+ * would bill the 768P tier against a 2K render, and `commit_credits` never
146
+ * collects the shortfall.
147
+ *
148
+ * Declare it in the catalog's own spelling. `normalizeVideoRequestParams`
149
+ * uses it in place of the nearest-band snap; `video-collapse-parity.test.ts`
150
+ * pins each declaration to what the provider normalizer actually returns, so
151
+ * the two cannot drift.
152
+ */
153
+ unlistedResolutionRendersAs?: string
136
154
  qualities?: readonly string[]
137
155
  durations?: readonly number[]
138
156
  pricing: readonly PriceVariant[]
@@ -155,6 +173,12 @@ export interface ModelCatalogEntry {
155
173
  * Frontend pickers ignore this flag.
156
174
  */
157
175
  mcpHidden?: boolean
176
+ /**
177
+ * The provider's safety filter is known to be non-deterministic on this
178
+ * model — the platform retries a blocked request once; `fallback` names
179
+ * the catalog model to offer when it blocks again.
180
+ */
181
+ safetyFilter?: { stochastic: true; fallback?: string }
158
182
  }
159
183
 
160
184
  /**
@@ -177,7 +201,7 @@ export const MODEL_RECOMMENDATIONS: readonly ModelRecommendation[] = [
177
201
  { intent: "cheapest realistic image", modelIds: ["z-image", "qwen", "imagen4-fast"], note: "Z-Image is the cheapest. Qwen / Imagen4 Fast for slightly higher quality." },
178
202
  { intent: "highest fidelity image", modelIds: ["nano-banana-pro", "imagen4-ultra", "flux-flex"], note: "Pick by family preference; all three are premium tiers." },
179
203
  { intent: "image edit / restyle", modelIds: ["flux-kontext", "ideogram-remix", "seedream-5-pro-i2i"], note: "Flux Kontext preserves identity; Ideogram Remix is character-aware; Seedream 5 Pro for instruction-based edits (5 Lite is the budget option)." },
180
- { intent: "highest-resolution image (4K / 8K)", modelIds: ["topaz-image-upscale", "nano-banana-pro", "gpt-image-2"], note: "Generate at native then Topaz upscale for 8K." },
204
+ { intent: "highest-resolution image", modelIds: ["topaz-image-upscale", "nano-banana-pro", "gpt-image-2"], note: "Generate at the model's top tier, then Topaz upscale 4x (Topaz's only lever is the 1x/2x/4x factor)." },
181
205
  { intent: "background removal / cutout", modelIds: ["recraft-remove-bg"], note: "Cheap, no prompt needed." },
182
206
  // video
183
207
  { intent: "best cinematic video", modelIds: ["veo3", "kling-3.0", "seedance-2"], note: "VEO 3.1 Quality for premium narrative; Kling 3.0 for music-synced motion; Seedance 2 for reference-driven consistency." },
@@ -545,6 +569,7 @@ const IMAGE_MODELS: Record<string, ModelCatalogEntry> = {
545
569
  { identifier: "gpt-image-2:2K", credits: 30, note: "2K" },
546
570
  { identifier: "gpt-image-2:4K", credits: 60, note: "4K" },
547
571
  ],
572
+ safetyFilter: { stochastic: true, fallback: "nano-banana-pro" },
548
573
  },
549
574
  "gpt-image-2-i2i": {
550
575
  id: "gpt-image-2-i2i",
@@ -975,14 +1000,15 @@ const IMAGE_MODELS: Record<string, ModelCatalogEntry> = {
975
1000
  family: "Topaz",
976
1001
  label: "Topaz Image Upscale",
977
1002
  series: "Topaz",
978
- description: "High-quality image upscale up to 8K. Best for production-ready output.",
1003
+ description: "High-quality image upscale at 1x (enhance only), 2x or 4x. Best for production-ready output.",
979
1004
  useCases: ["upscale", "high-res", "premium"],
980
1005
  features: ["reference-image"],
981
- resolutions: ["2K", "4K", "8K"],
1006
+ // No `resolutions`: the provider's only quality lever is `upscale_factor`
1007
+ // (1/2/4) — see resolveTopazUpscale. The 2K/4K/8K menu this used to
1008
+ // advertise had no provider parameter behind it.
982
1009
  pricing: [
983
- { identifier: "topaz-image-upscale", credits: 25, note: "2K default" },
984
- { identifier: "topaz-image-upscale:4K", credits: 50, note: "4K" },
985
- { identifier: "topaz-image-upscale:8K", credits: 100, note: "8K" },
1010
+ { identifier: "topaz-image-upscale", credits: 25, note: "1x / 2x" },
1011
+ { identifier: "topaz-image-upscale:4K", credits: 50, note: "4x (legacy identifier)" },
986
1012
  ],
987
1013
  },
988
1014
  }
@@ -1017,6 +1043,7 @@ const VIDEO_MODELS: Record<string, ModelCatalogEntry> = {
1017
1043
  // docs.kie.ai/market/minimax-h3/{text,image,reference}-to-video.
1018
1044
  "minimax-h3": {
1019
1045
  id: "minimax-h3",
1046
+ unlistedResolutionRendersAs: "2K",
1020
1047
  kind: "video",
1021
1048
  modes: ["i2v", "t2v"] as const,
1022
1049
  family: "MiniMax",
@@ -1505,6 +1532,7 @@ const VIDEO_MODELS: Record<string, ModelCatalogEntry> = {
1505
1532
  // declared explicitly in PRICING_DEFAULT_RESOLUTION, never by array position.
1506
1533
  "wan-3": {
1507
1534
  id: "wan-3",
1535
+ unlistedResolutionRendersAs: "720p",
1508
1536
  kind: "video",
1509
1537
  modes: ["i2v", "t2v"] as const,
1510
1538
  family: "Alibaba",
@@ -1529,6 +1557,7 @@ const VIDEO_MODELS: Record<string, ModelCatalogEntry> = {
1529
1557
  },
1530
1558
  "wan-3-prime": {
1531
1559
  id: "wan-3-prime",
1560
+ unlistedResolutionRendersAs: "720p",
1532
1561
  kind: "video",
1533
1562
  modes: ["i2v", "t2v"] as const,
1534
1563
  family: "Alibaba",
@@ -1697,6 +1726,76 @@ const VIDEO_MODELS: Record<string, ModelCatalogEntry> = {
1697
1726
  { identifier: "wan-2.7-t2v", credits: 188, note: "5s 720p default" },
1698
1727
  ],
1699
1728
  },
1729
+ // Lightricks LTX 2.3 (Replicate, not KIE). Both variants run one endpoint per
1730
+ // variant and switch behaviour with a `task` discriminator, so t2v and i2v are
1731
+ // the SAME id — no VIDEO_MODE_ALIASES row. `extend` / `retake` are separate
1732
+ // priced ops on the same model (ids below) but are NOT listed in `modes`:
1733
+ // they are driven by the Extend Video / Video Retake nodes' own pickers, and
1734
+ // adding the mode here would duplicate LTX in those menus.
1735
+ // Bands are lowercase to match LTX_DURATION_TIERS' keys in credit-identifiers.ts
1736
+ // and QUALITY_MAP in node-default-mappings.ts.
1737
+ "ltx-2.3-pro": {
1738
+ id: "ltx-2.3-pro",
1739
+ kind: "video",
1740
+ modes: ["i2v", "t2v"] as const,
1741
+ family: "Lightricks",
1742
+ label: "LTX 2.3 Pro",
1743
+ series: "LTX",
1744
+ description: "Lightricks LTX 2.3 Pro — text/image/audio→video up to 4K, 6/8/10s, end-frame interpolation.",
1745
+ useCases: ["premium", "high-res", "narrative"],
1746
+ features: ["end-frame"],
1747
+ aspectRatios: ["16:9", "9:16"] as const,
1748
+ resolutions: ["1080p", "2k", "4k"] as const,
1749
+ durations: [6, 8, 10],
1750
+ pricing: [
1751
+ { identifier: "ltx-2.3-pro", credits: 240, note: "default 1080p 6s" },
1752
+ { identifier: "ltx-2.3-pro:1080p:6s", credits: 240 },
1753
+ { identifier: "ltx-2.3-pro:1080p:8s", credits: 320 },
1754
+ { identifier: "ltx-2.3-pro:1080p:10s", credits: 400 },
1755
+ { identifier: "ltx-2.3-pro:2k:6s", credits: 480 },
1756
+ { identifier: "ltx-2.3-pro:2k:8s", credits: 640 },
1757
+ { identifier: "ltx-2.3-pro:2k:10s", credits: 800 },
1758
+ { identifier: "ltx-2.3-pro:4k:6s", credits: 960 },
1759
+ { identifier: "ltx-2.3-pro:4k:8s", credits: 1280 },
1760
+ { identifier: "ltx-2.3-pro:4k:10s", credits: 1600 },
1761
+ { identifier: "ltx-2.3-pro-extend:per-second", credits: 40, note: "extend, per second of new footage" },
1762
+ { identifier: "ltx-2.3-pro-retake:per-second", credits: 40, note: "retake, per second re-rendered" },
1763
+ ],
1764
+ },
1765
+ "ltx-2.3-fast": {
1766
+ id: "ltx-2.3-fast",
1767
+ kind: "video",
1768
+ modes: ["i2v", "t2v"] as const,
1769
+ family: "Lightricks",
1770
+ label: "LTX 2.3 Fast",
1771
+ series: "LTX",
1772
+ description: "Lightricks LTX 2.3 Fast — text/image→video up to 20s at 1080p (6/8/10s at 2K and 4K). No audio input, no extend.",
1773
+ useCases: ["long-form", "fast", "narrative"],
1774
+ features: ["end-frame"],
1775
+ aspectRatios: ["16:9", "9:16"] as const,
1776
+ resolutions: ["1080p", "2k", "4k"] as const,
1777
+ // Flat union across bands — 12–20s exist only at 1080p (LTX_DURATION_TIERS
1778
+ // in credit-identifiers.ts is the per-band authority and snaps a 2k/4k
1779
+ // request back onto 6/8/10s, so the reservation is always a real tier).
1780
+ durations: [6, 8, 10, 12, 14, 16, 18, 20],
1781
+ pricing: [
1782
+ { identifier: "ltx-2.3-fast", credits: 180, note: "default 1080p 6s" },
1783
+ { identifier: "ltx-2.3-fast:1080p:6s", credits: 180 },
1784
+ { identifier: "ltx-2.3-fast:1080p:8s", credits: 240 },
1785
+ { identifier: "ltx-2.3-fast:1080p:10s", credits: 300 },
1786
+ { identifier: "ltx-2.3-fast:1080p:12s", credits: 360 },
1787
+ { identifier: "ltx-2.3-fast:1080p:14s", credits: 420 },
1788
+ { identifier: "ltx-2.3-fast:1080p:16s", credits: 480 },
1789
+ { identifier: "ltx-2.3-fast:1080p:18s", credits: 540 },
1790
+ { identifier: "ltx-2.3-fast:1080p:20s", credits: 600 },
1791
+ { identifier: "ltx-2.3-fast:2k:6s", credits: 360 },
1792
+ { identifier: "ltx-2.3-fast:2k:8s", credits: 480 },
1793
+ { identifier: "ltx-2.3-fast:2k:10s", credits: 600 },
1794
+ { identifier: "ltx-2.3-fast:4k:6s", credits: 720 },
1795
+ { identifier: "ltx-2.3-fast:4k:8s", credits: 960 },
1796
+ { identifier: "ltx-2.3-fast:4k:10s", credits: 1200 },
1797
+ ],
1798
+ },
1700
1799
  // ── HappyHorse (1.1 — ids kept version-less; repointed in place when KIE
1701
1800
  // delisted 1.0, same param surface, so saved workflows keep working) ──
1702
1801
  "happyhorse": {
@@ -2714,6 +2813,187 @@ export function normalizeModelInput(
2714
2813
  return out
2715
2814
  }
2716
2815
 
2816
+ /** Aspect tokens that mean "match the input" or "let the provider decide". They
2817
+ * are NOT concrete ratios, so snapping them onto one is always wrong — the
2818
+ * provider adapter (applySeedance2Params, applyMinimaxH3Params) resolves them
2819
+ * once it knows the run's mode. */
2820
+ const PASSTHROUGH_ASPECT_TOKENS = new Set(["auto", "adaptive"])
2821
+
2822
+ /** Approximate vertical pixel count per band token, for nearest-band matching.
2823
+ * Anything of the form `<N>p` is parsed directly; these are the named bands
2824
+ * the catalogs use that do not follow that form. */
2825
+ const RESOLUTION_BAND_PIXELS: Record<string, number> = {
2826
+ "2k": 1440,
2827
+ "4k": 2160,
2828
+ "8k": 4320,
2829
+ }
2830
+
2831
+ function bandPixels(token: string): number | undefined {
2832
+ const t = token.trim().toLowerCase()
2833
+ if (RESOLUTION_BAND_PIXELS[t] !== undefined) return RESOLUTION_BAND_PIXELS[t]
2834
+ const m = /^(\d+)p$/.exec(t)
2835
+ return m ? Number(m[1]) : undefined
2836
+ }
2837
+
2838
+ /** Nearest member of `allowed` to `token` by vertical pixels. An UNPARSEABLE
2839
+ * request falls back to the highest declared band — never the cheapest, which
2840
+ * is what a catalog-order fallback would give (R7). Highest is computed from
2841
+ * the pixel counts, not from the array position: most video catalogs list
2842
+ * ascending but not all do (minimax-h3 is `["2K", "768P"]`), and an ordering
2843
+ * convention is not something a money path should depend on. */
2844
+ function nearestResolutionBand(token: string, allowed: readonly string[]): string {
2845
+ let highest = allowed[allowed.length - 1]!
2846
+ let highestPx = -Infinity
2847
+ for (const a of allowed) {
2848
+ const px = bandPixels(a)
2849
+ if (px !== undefined && px > highestPx) { highestPx = px; highest = a }
2850
+ }
2851
+ const want = bandPixels(token)
2852
+ if (want === undefined) return highest
2853
+ let best = highest
2854
+ let bestDist = Infinity
2855
+ for (const a of allowed) {
2856
+ const px = bandPixels(a)
2857
+ if (px === undefined) continue
2858
+ const d = Math.abs(px - want)
2859
+ if (d < bestDist) { bestDist = d; best = a }
2860
+ }
2861
+ return best
2862
+ }
2863
+
2864
+ /** Nearest aspect ratio in log space. Mirrors
2865
+ * `backend/src/providers/video/aspect-ratio.ts` exactly — that helper cannot
2866
+ * be imported here (wrong package), and the totality test in
2867
+ * `backend/src/lib/mcp/__tests__/video-normalizer-totality.test.ts` pins the
2868
+ * two implementations to the same answer for every catalogued model. */
2869
+ function nearestAspectRatio(token: string, allowed: readonly string[]): string | undefined {
2870
+ const [w, h] = token.split(":").map(Number)
2871
+ if (!w || !h) return undefined
2872
+ const target = Math.log(w / h)
2873
+ let best: string | undefined
2874
+ let bestDist = Infinity
2875
+ for (const c of allowed) {
2876
+ const [cw, ch] = c.split(":").map(Number)
2877
+ if (!cw || !ch) continue
2878
+ const d = Math.abs(Math.log(cw / ch) - target)
2879
+ if (d < bestDist) { bestDist = d; best = c }
2880
+ }
2881
+ return best
2882
+ }
2883
+
2884
+ /**
2885
+ * Read a catalog lever off an UNVALIDATED input as a trimmed string.
2886
+ *
2887
+ * `normalizeVideoRequestParams` runs in both routes' creditGuard preHandler —
2888
+ * which fires BEFORE the route's Zod parse — and in all four `buildPayload`
2889
+ * branches, whose `data` is persisted workflow JSON that an agent/import/
2890
+ * FieldMapping can write any JSON type into. A bare `.trim()` therefore throws
2891
+ * `TypeError` on `{"resolution": 1080}`: a 500 where the route used to return a
2892
+ * clean Zod 400, and in the DAG a thrown lever takes the WHOLE run down after
2893
+ * sibling nodes have already reserved and generated. So coerce rather than
2894
+ * assume — same posture as `sameOptionValue`, which has always used `String(v)`.
2895
+ *
2896
+ * `null`/`undefined`/blank mean "absent" (there is nothing to snap) and read as
2897
+ * `undefined`, so the priced fill downstream supplies the band the identifier
2898
+ * assumes — price and wire still agree. Everything else is stringified: a
2899
+ * numeric `1080` becomes `"1080"` and snaps like its string form.
2900
+ */
2901
+ function readOptionToken(value: unknown): string | undefined {
2902
+ if (value === undefined || value === null) return undefined
2903
+ const s = (typeof value === "string" ? value : String(value)).trim()
2904
+ return s === "" ? undefined : s
2905
+ }
2906
+
2907
+ export interface NormalizedVideoRequest {
2908
+ aspectRatio?: string
2909
+ resolution?: string
2910
+ adjustments: ModelInputAdjustment[]
2911
+ }
2912
+
2913
+ /**
2914
+ * The video lane's LIMITED normalizer: `aspectRatio` + `resolution` only, and
2915
+ * only when the catalog actually declares the option list.
2916
+ *
2917
+ * Three deliberate differences from `normalizeModelInput`:
2918
+ * 1. It never DROPS a lever. `normalizeModelInput` removes a value whose list
2919
+ * the catalog doesn't declare; on the video lane `resolution` is a pricing
2920
+ * input, so dropping it would lower the reserved tier and `commit_credits`
2921
+ * never collects an upward delta — a params fix would become a money bug.
2922
+ * 2. "Auto" / "adaptive" pass through (see PASSTHROUGH_ASPECT_TOKENS).
2923
+ * 3. It snaps to the NEAREST option, not to `allowed[0]`. `normalizeModelInput`
2924
+ * would turn a portrait 9:21 into landscape 16:9 and a 4k request into the
2925
+ * cheapest 480p; nearest matches what the provider adapters and the MCP
2926
+ * normalizer already do, so all three give ONE answer. Its snap policy is
2927
+ * deliberately left alone — it is shared with the image lane and the
2928
+ * copilot, and PR 4 defers changing it.
2929
+ *
2930
+ * `duration` is deliberately NOT normalized here: the LTX bands make duration
2931
+ * legality resolution-dependent, which a flat catalog list can't express, and
2932
+ * `buildVideoCreditModelIdentifier` already snaps duration onto a seeded tier.
2933
+ * Carrying THAT tier to the wire is `pricedVideoSelection`'s job
2934
+ * (`credit-identifiers.ts`), which runs after this one, once resolution is known.
2935
+ *
2936
+ * An OMITTED lever stays omitted here — filling it is a pricing decision, not a
2937
+ * catalog one, and likewise belongs to `pricedVideoSelection`.
2938
+ *
2939
+ * Case is canonicalised to the catalog's own spelling — the credit identifiers
2940
+ * key their tables case-SENSITIVELY (LTX_DURATION_TIERS), so this is the single
2941
+ * place a "4K" becomes "4k", and it runs before every identifier site.
2942
+ *
2943
+ * Pure — so the credit-identifier preHandler, `computeCredits`, the reservation
2944
+ * site and the payload build can all call it and cannot disagree (spec §4).
2945
+ */
2946
+ export function normalizeVideoRequestParams(
2947
+ modelId: string,
2948
+ input: { aspectRatio?: string; resolution?: string },
2949
+ ): NormalizedVideoRequest {
2950
+ const m = MODEL_CATALOG[modelId]
2951
+ // Seeded from the COERCED tokens, never the raw input: this function's own
2952
+ // return type promises `string | undefined`, and its callers feed it straight
2953
+ // to the credit identifier and the provider wire. Handing back the caller's
2954
+ // `1080` or `[]` verbatim would break that promise at exactly the sites that
2955
+ // must agree on one value.
2956
+ const aspect = readOptionToken(input.aspectRatio)
2957
+ const res = readOptionToken(input.resolution)
2958
+ const out: NormalizedVideoRequest = { aspectRatio: aspect, resolution: res, adjustments: [] }
2959
+ if (!m) return out
2960
+
2961
+ if (m.aspectRatios?.length && aspect && !PASSTHROUGH_ASPECT_TOKENS.has(aspect.toLowerCase())) {
2962
+ const allowed = m.aspectRatios
2963
+ const exact = allowed.find((a) => a === aspect) ?? allowed.find((a) => a.toLowerCase() === aspect.toLowerCase())
2964
+ if (exact !== undefined) {
2965
+ out.aspectRatio = exact
2966
+ } else {
2967
+ const next = nearestAspectRatio(aspect, allowed.filter((a) => !PASSTHROUGH_ASPECT_TOKENS.has(a.toLowerCase()))) ?? allowed[0]!
2968
+ out.adjustments.push({
2969
+ field: "aspectRatio", from: aspect, to: next,
2970
+ reason: `${m.label} does not support aspect ratio "${aspect}" — using "${next}" instead. Supported: ${allowed.join(", ")}.`,
2971
+ })
2972
+ out.aspectRatio = next
2973
+ }
2974
+ }
2975
+
2976
+ if (m.resolutions?.length && res) {
2977
+ const allowed = m.resolutions
2978
+ const exact = allowed.find((a) => a === res) ?? allowed.find((a) => a.toLowerCase() === res.toLowerCase())
2979
+ if (exact !== undefined) {
2980
+ out.resolution = exact
2981
+ } else {
2982
+ // A provider that COLLAPSES an unrecognised band renders its declared
2983
+ // default no matter what we send, so that default — not the nearest band
2984
+ // — is the value the wire and the price must both carry.
2985
+ const next = m.unlistedResolutionRendersAs ?? nearestResolutionBand(res, allowed)
2986
+ out.adjustments.push({
2987
+ field: "resolution", from: res, to: next,
2988
+ reason: `${m.label} does not support resolution "${res}" — using "${next}" instead. Supported: ${allowed.join(", ")}.`,
2989
+ })
2990
+ out.resolution = next
2991
+ }
2992
+ }
2993
+
2994
+ return out
2995
+ }
2996
+
2717
2997
  // =============================================================================
2718
2998
  // Frontend picker helpers — return `{value, label}[]` shapes that the
2719
2999
  // existing config-panel components expect, derived from the catalog so we
@@ -2742,6 +3022,7 @@ const MODEL_VALUE_LABELS: Record<string, string> = {
2742
3022
  "2K": "2K (High)",
2743
3023
  "4K": "4K (Ultra)",
2744
3024
  "4k": "4K",
3025
+ "2k": "2K",
2745
3026
  "8K": "8K (Ultra)",
2746
3027
  // qualities
2747
3028
  "medium": "Medium (Balanced)",
@@ -13,6 +13,42 @@ export const CREDIT_BASE_USD = 0.002
13
13
  * schemas and the factory-preset guard test all read this. Prompt cap ONLY (not negativePrompt). */
14
14
  export const IMAGE_PROMPT_MAX = 5000
15
15
 
16
+ /**
17
+ * The ONE aspect-ratio vocabulary the image routes' Zod enums are built from —
18
+ * `/v1/generate-image`, `/v1/image-to-image` and `/v1/edit-image` all declare
19
+ * `z.enum(IMAGE_ASPECT_RATIO_VALUES)` instead of keeping three literal lists
20
+ * that drift.
21
+ *
22
+ * It is the UNION of every ratio any `kind: "image"` catalog entry declares, so
23
+ * a value the picker offers can never 400 at the route. That gap has shipped
24
+ * twice — Wan 2.7's ultra-wide `8:1`/`1:8` (fixed on generate-image only) and
25
+ * Nano Banana 2 Lite's banner `4:1`/`1:4` (still open until this tuple landed).
26
+ * `packages/shared/src/__tests__/image-pricing-catalog-coverage.test.ts` fails
27
+ * the build if a new model declares a ratio this tuple is missing.
28
+ *
29
+ * This enum is a VOCABULARY BOUND, not a per-model gate: it only stops a
30
+ * free-form string from reaching a provider as `image_size`. The per-model gate
31
+ * is the catalog snap (`resolveNormalizedImageGen` → `normalizeModelInput`),
32
+ * which CORRECTS an unsupported ratio and discloses it in the response's
33
+ * `adjustments` — never rejects. So widening this tuple is always safe: it
34
+ * defers a rejection to the correcting snap, which is the desired behaviour.
35
+ *
36
+ * `auto` is model-specific (GPT Image 2 / Nano Banana 2 Lite treat it as
37
+ * "native"); it lives here for the same reason as the rest — the snap drops or
38
+ * replaces it for models that do not declare it.
39
+ */
40
+ export const IMAGE_ASPECT_RATIO_VALUES = [
41
+ "auto",
42
+ "1:1", "16:9", "9:16", "4:3", "3:4",
43
+ "3:2", "2:3", "5:4", "4:5", "21:9",
44
+ // Ultra-wide / ultra-tall banner ratios: Wan 2.7 + Wan 2.7 Pro (8:1, 1:8)
45
+ // and Nano Banana 2 Lite (4:1, 1:4, 8:1, 1:8).
46
+ "4:1", "1:4", "8:1", "1:8",
47
+ ] as const
48
+
49
+ /** A ratio string the image routes accept (pre-snap vocabulary, not a per-model guarantee). */
50
+ export type ImageAspectRatio = typeof IMAGE_ASPECT_RATIO_VALUES[number]
51
+
16
52
  /**
17
53
  * Per-provider maximum ASSEMBLED image-prompt length (chars), VERIFIED against
18
54
  * each model's official docs.kie.ai schema (2026-06). Providers absent here use
@@ -75,15 +111,37 @@ export const VIDEO_PROMPT_MAX = 8000
75
111
 
76
112
  /**
77
113
  * Suno prompt / lyrics / content ceiling — the LARGEST any Suno version accepts
78
- * in custom mode (V4.5 / V4.5PLUS / V4.5ALL / V5 / V5.5 = 5000). The route Zod
79
- * uses this as a generous ceiling; the handler clamps to the per-version cap via
80
- * {@link getMaxSunoPromptChars} (V4/V3.5 = 3000, non-custom = 500). Shared with
81
- * the editor `maxLength` / counter (warn-don't-block at the per-version cap).
82
- * `style` and `title` have their own caps ({@link getMaxSunoStyleChars} /
83
- * {@link SUNO_TITLE_MAX}).
114
+ * in custom mode (V4.5 / V4.5PLUS / V4.5ALL / V5 / V5.5 = 5000). NOT the route
115
+ * Zod bound — the routes bind at {@link SUNO_HARD_CEILING} and clamp to the
116
+ * per-version cap via {@link getMaxSunoPromptChars} (3000 non-custom for every
117
+ * version; in custom mode 3000 for V4/V3.5, 5000 for V4.5+/V5) before the job
118
+ * is built. This constant is shared with the editor `maxLength` / counter
119
+ * (warn-don't-block at the per-version cap). `style` and `title` have their
120
+ * own caps ({@link getMaxSunoStyleChars} / {@link SUNO_TITLE_MAX}).
84
121
  */
85
122
  export const SUNO_TEXT_MAX = 5000
86
123
 
124
+ /**
125
+ * Absolute ceiling for every Suno text field on the route Zod schemas
126
+ * (`prompt`, `lyrics`, `fullLyrics`, `content`, `userPrompt`).
127
+ *
128
+ * WARN-DON'T-BLOCK. The per-version caps ({@link getMaxSunoPromptChars} /
129
+ * {@link getMaxSunoStyleChars}) do the real work — the routes CLAMP to them
130
+ * before building the job. The Zod bound only exists to stop abuse, so it must
131
+ * stay generous: until 2026-09 the routes bounded these fields at
132
+ * {@link SUNO_TEXT_MAX} (5000), which meant a programmatically-set prompt (an
133
+ * agent, an app run, a FieldMapping) was hard-REJECTED with a 400 before the
134
+ * clamp could trim it — three app-report rows on 2026-08-19..31.
135
+ *
136
+ * DISTINCT FROM {@link PROMPT_HARD_CEILING} on purpose: that one is an
137
+ * image/video budget pinned to MAX_IMAGE/VIDEO_PROMPT_CHARS_BY_PROVIDER by its
138
+ * own drift guard. Coupling them would let one modality's limits move the
139
+ * other's. This one has a single invariant, enforced in
140
+ * `__tests__/prompt-length-limits.test.ts`: it MUST stay >= every value
141
+ * getMaxSunoPromptChars / getMaxSunoStyleChars can return.
142
+ */
143
+ export const SUNO_HARD_CEILING = 30000
144
+
87
145
  /**
88
146
  * Absolute ceiling for the `prompt` / `negativePrompt` fields on the image and
89
147
  * video routes' Zod schemas. The PER-MODEL limits below (and the editor warning)
@@ -238,9 +296,11 @@ export function getMaxTtsChars(provider: string | undefined): number {
238
296
  }
239
297
 
240
298
  /**
241
- * Suno per-version field caps (from docs.kie.ai/suno-api/generate-music). The old
242
- * flat {@link SUNO_TEXT_MAX} (3000) was simultaneously too low for V4.5+/V5
243
- * prompts (5000) and too high for `style` (1000) and `title` (80).
299
+ * Suno per-version field caps (from docs.kie.ai/suno-api/generate-music). The
300
+ * original flat 3000 cap was simultaneously too low for V4.5+/V5 prompts
301
+ * (5000) and too high for `style` (1000) and `title` (80). ({@link
302
+ * SUNO_TEXT_MAX} is 5000 today — the largest per-version prompt cap; the
303
+ * route Zod bound is {@link SUNO_HARD_CEILING}.)
244
304
  * - prompt / lyrics: 3000 in non-custom mode (all versions); in custom mode
245
305
  * 3000 for V4/V3.5 and 5000 for V4.5 / V4.5PLUS / V4.5ALL / V5 / V5.5.
246
306
  * - style: 200 for V4/V3.5, 1000 for V4.5+.
@@ -1840,6 +1900,89 @@ export const VIDEO_REF_LIMITS_BY_PROVIDER: Record<
1840
1900
  // verified provider path + the catalog `reference-image` feature.
1841
1901
  }
1842
1902
 
1903
+ /** Per-reference-video duration bounds, in seconds. */
1904
+ export interface RefVideoDurationLimit {
1905
+ minSec: number
1906
+ maxSec: number
1907
+ /** Cap on the SUM of all reference-video durations, when the provider has one. */
1908
+ maxTotalSec?: number
1909
+ }
1910
+
1911
+ /**
1912
+ * Reference-video duration limits, provider-declared.
1913
+ *
1914
+ * `VIDEO_REF_LIMITS_BY_PROVIDER` above caps the reference COUNT; these cap the
1915
+ * duration of each clip. Both routes already ffprobe every reference video to
1916
+ * price the run (ee/billing/seedance2-ref-video-credits.ts and
1917
+ * minimax-h3-credits.ts), so the numbers are in hand before the job exists —
1918
+ * and until 2026-09-02 they were discarded, which is why "Each reference video
1919
+ * must be between 2 and 30 seconds" and "video duration 52838 ms, expected
1920
+ * [2000, 15000] ms" were reaching users as post-payment provider rejects
1921
+ * (app-reports §11.3).
1922
+ *
1923
+ * Data-driven, exactly like SEEDANCE_2_R2V_MAX_AUDIO_SEC_BY_PROVIDER above:
1924
+ * only providers with a VERIFIED documented limit are listed, so an unknown
1925
+ * provider is never false-rejected.
1926
+ *
1927
+ * Sources: docs.kie.ai/market/bytedance/seedance-2-5 ("Single video duration:
1928
+ * [2, 30] seconds"; "Total duration of reference videos must not exceed 30
1929
+ * seconds") and docs.kie.ai/market/minimax-h3/reference-to-video (reference
1930
+ * videos are "2 to 15 seconds" each and "the total duration of all reference
1931
+ * videos cannot exceed 15 seconds", up to 3 videos) — both fetched 2026-09-02.
1932
+ */
1933
+ export const VIDEO_REF_VIDEO_DURATION_LIMITS: Record<string, RefVideoDurationLimit | undefined> = {
1934
+ "seedance-2-5": { minSec: 2, maxSec: 30, maxTotalSec: 30 },
1935
+ // MiniMax Hailuo 3 — the provider states the per-clip bound in its own reject
1936
+ // text: "content[1].video_url: invalid param: video duration 52838 ms,
1937
+ // expected [2000, 15000] ms" (app-reports P4, 2 rows, the same clip retried).
1938
+ // The KIE reference-to-video doc adds the COMBINED cap — the same shape the
1939
+ // audio side already declares in SEEDANCE_2_R2V_MAX_AUDIO_SEC_BY_PROVIDER
1940
+ // above, whose h3 row cites the same page.
1941
+ "minimax-h3": { minSec: 2, maxSec: 15, maxTotalSec: 15 },
1942
+ }
1943
+
1944
+ /**
1945
+ * Validate probed reference-video durations against the provider's limits.
1946
+ *
1947
+ * Non-finite / non-positive entries are IGNORED, never rejected: the probe
1948
+ * helpers treat an unreadable clip as a worst-case duration for pricing, and
1949
+ * turning a failed ffprobe into a user-facing 400 would block legitimate runs
1950
+ * on a transient probe failure. Callers therefore pass the RAW per-URL probe
1951
+ * outcomes (a rejected ffprobe arrives here as NaN) — an ignored entry also
1952
+ * contributes nothing to the `maxTotalSec` sum, so a probe blip can never push
1953
+ * an otherwise-legal set over the total cap either.
1954
+ */
1955
+ export function checkRefVideoDurations(
1956
+ provider: string,
1957
+ durationsSec: readonly number[],
1958
+ ): { ok: true } | { ok: false; message: string } {
1959
+ const limit = VIDEO_REF_VIDEO_DURATION_LIMITS[provider]
1960
+ if (!limit) return { ok: true }
1961
+
1962
+ const usable = durationsSec.filter((d) => Number.isFinite(d) && d > 0)
1963
+ if (usable.length === 0) return { ok: true }
1964
+
1965
+ const offender = usable.find((d) => d < limit.minSec || d > limit.maxSec)
1966
+ if (offender !== undefined) {
1967
+ return {
1968
+ ok: false,
1969
+ message: `Each reference video must be between ${limit.minSec} and ${limit.maxSec} seconds — one is ${offender.toFixed(1)}s. Trim it (a Trim Video node upstream works) and run again.`,
1970
+ }
1971
+ }
1972
+
1973
+ if (limit.maxTotalSec !== undefined) {
1974
+ const total = usable.reduce((a, b) => a + b, 0)
1975
+ if (total > limit.maxTotalSec) {
1976
+ return {
1977
+ ok: false,
1978
+ message: `Reference videos must not exceed ${limit.maxTotalSec} seconds in total — these add up to ${total.toFixed(1)}s. Remove one or trim them and run again.`,
1979
+ }
1980
+ }
1981
+ }
1982
+
1983
+ return { ok: true }
1984
+ }
1985
+
1843
1986
  /**
1844
1987
  * Video models where credit cost depends on resolution AND whether a video
1845
1988
  * reference is connected. Identifier suffix: `:{resolution}[-ref]`.
@@ -2027,8 +2170,8 @@ export interface VideoAudioCapability {
2027
2170
  /**
2028
2171
  * Per-model audio capability. Only models that produce SOME audio are listed;
2029
2172
  * anything absent defaults to `{ mode: "none" }` via `getVideoAudioCapability`,
2030
- * so silent models (minimax, hailuo, wan, grok-i2v, gemini-omni-video, runway,
2031
- * pika, …) need no entry. New audio-capable models MUST be added here — the
2173
+ * so silent models (minimax, hailuo, wan-i2v, grok-i2v, runway, pika, …) need
2174
+ * no entry. New audio-capable models MUST be added here — the
2032
2175
  * `video-audio-capability` guard test cross-checks this map against the model
2033
2176
  * configs' `extraParams.sound` / `generate_audio` + VEO/Seedance-2 sets so a
2034
2177
  * forgotten entry fails CI rather than silently disabling audio.
@@ -2077,10 +2220,39 @@ export const VIDEO_AUDIO_CAPABILITY: Record<string, VideoAudioCapability> = {
2077
2220
  // and skip the lip-sync pass. `defaultOn` mirrors the KIE default so an
2078
2221
  // intent-less request is described honestly; audio is priced into the uniform
2079
2222
  // per-second rate, so NOT cost-affecting (no `:audio` composite).
2080
- // gemini-omni-flash is deliberately ABSENT — gemini-omni-video is absent too
2081
- // (mode "none"), and the siblings must not disagree.
2082
2223
  "wan-3": { mode: "ambient", field: "audio", defaultOn: true },
2083
2224
  "wan-3-prime": { mode: "ambient", field: "audio", defaultOn: true },
2225
+ // Gemini Omni (both SKUs — the pro `gemini-omni-video` and the faster
2226
+ // `gemini-omni-flash`; one model at two speeds, so their rows are identical
2227
+ // and a test pins that). Settled 2026-09-03 from Google's own documentation
2228
+ // at https://ai.google.dev/gemini-api/docs/omni, having been unlisted — and
2229
+ // therefore reported as SILENT — while the catalog described both as "native
2230
+ // audio".
2231
+ //
2232
+ // "ambient", and alwaysOn with no toggle field, on three sentences from that
2233
+ // page:
2234
+ // - "By default the model will try to generate an appropriate audio track
2235
+ // for a video." Audio on every render, and the KIE input schema
2236
+ // (prompt / image_urls / first+last_frame_url / audio_ids / video_list /
2237
+ // character_ids / duration / aspect_ratio / seed / resolution — see
2238
+ // docs.kie.ai/market/gemini-omni-video and
2239
+ // docs.kie.ai/market/google/gemini-omni-flash-1-1) carries no on/off
2240
+ // lever, so there is nothing for applyVideoAudioToggle to write.
2241
+ // - NOT native_speech: "Multi-turn voice extension: Generating spoken
2242
+ // dialogue or speech is supported when extending previously generated
2243
+ // videos via multi-turn (`previous_interaction_id`)" — a field KIE's
2244
+ // createTask schema does not expose, so the dialogue path is unreachable
2245
+ // on our transport. Held to the Wan 3.0 bar: no documented dialogue
2246
+ // guarantee on the path we actually drive ⇒ ambient, and upgrade only on
2247
+ // a live probe (the kling-3.0 standard). Classifying it native_speech
2248
+ // would reroute the Story→Video dialogue pipeline past the lip-sync pass.
2249
+ // - NOT audio_driven either: "Uploading audio references is unsupported in
2250
+ // the current version of the API", and "any audio in a video reference is
2251
+ // ignored" — there is no reference-audio transport to be driven by, and
2252
+ // runGeminiOmni never sends `audio_ids`.
2253
+ // Audio is priced into the per-tier rate, so NOT cost-affecting.
2254
+ "gemini-omni-video": { mode: "ambient", alwaysOn: true },
2255
+ "gemini-omni-flash": { mode: "ambient", alwaysOn: true },
2084
2256
  }
2085
2257
 
2086
2258
  const VIDEO_AUDIO_NONE: VideoAudioCapability = { mode: "none" }