keating 4.0.2 → 4.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/keating.js +1 -1
- package/dist/src/core/version.js +1 -1
- package/package.json +1 -1
- package/web/.output/nitro.json +1 -1
- package/web/.output/public/assets/{AnimationPlayer-Bj9E6XAF.js → AnimationPlayer-CYbhLL_R.js} +1 -1
- package/web/.output/public/assets/{AssistantChatPanel-CPiSdBQo.js → AssistantChatPanel-DQSKpGWP.js} +122 -122
- package/web/.output/public/assets/{CategoricalChart-ZCOId7ho.js → CategoricalChart-BVb9-7_N.js} +1 -1
- package/web/.output/public/assets/{Chat-BPuDBuv7.js → Chat-C6p8yxAr.js} +2 -2
- package/web/.output/public/assets/{CodeHighlighter-CUd-nVJG.js → CodeHighlighter-DhI0IV5H.js} +1 -1
- package/web/.output/public/assets/{ComingUp-BM3f_QSf.js → ComingUp-DFaRgEbt.js} +1 -1
- package/web/.output/public/assets/{CourseJoin--37EPPeg.js → CourseJoin-DMZ6foZO.js} +1 -1
- package/web/.output/public/assets/CourseWorkspace-l_bPGNhE.js +25 -0
- package/web/.output/public/assets/{Courses-DgryGTWV.js → Courses-DIqulPB-.js} +1 -1
- package/web/.output/public/assets/{EvolutionDetail-BHwyObAD.js → EvolutionDetail-C-NYL1qL.js} +1 -1
- package/web/.output/public/assets/{FlashcardRenderer-hzqoLbnQ.js → FlashcardRenderer-iDgfcwk4.js} +1 -1
- package/web/.output/public/assets/{FlashcardShaderField-pDI9y_r4.js → FlashcardShaderField-JGgL2kgp.js} +1 -1
- package/web/.output/public/assets/{FlashcardTimer-BANuXJze.js → FlashcardTimer-vPNL9_0N.js} +1 -1
- package/web/.output/public/assets/{HyperframesPlayer-BQFYx43N.js → HyperframesPlayer-DEtdwGBR.js} +1 -1
- package/web/.output/public/assets/KeatingBench-XnWxgfUC.js +2 -0
- package/web/.output/public/assets/{LearningInsightsHeader-Do3bROTu.js → LearningInsightsHeader-DWj2-1xb.js} +1 -1
- package/web/.output/public/assets/{Live-HZjRDq-V.js → Live-BasKZMuE.js} +1 -1
- package/web/.output/public/assets/{MarkdownBlock-KUcDESQo.js → MarkdownBlock-C89ecD-M.js} +3 -3
- package/web/.output/public/assets/ModelSelector-B_EaHvVz.js +1 -0
- package/web/.output/public/assets/{Nav-GcfPhZc3.js → Nav-DQ-5KEsA.js} +1 -1
- package/web/.output/public/assets/{NotOrganicAccessPromptDialog-CKb6iTFn.js → NotOrganicAccessPromptDialog-DS7PLMx4.js} +1 -1
- package/web/.output/public/assets/{NotOrganicCallback-DWbhlP2j.js → NotOrganicCallback-BxKOsisT.js} +1 -1
- package/web/.output/public/assets/{OAuthCallback-D15ohKxE.js → OAuthCallback-dk2EQnAU.js} +1 -1
- package/web/.output/public/assets/OpenUIPreview-BrBXAEt4.js +1 -0
- package/web/.output/public/assets/PatternDigestPanel-jmxcLCUX.js +1 -0
- package/web/.output/public/assets/Plus-RIb_ldNd.js +1 -0
- package/web/.output/public/assets/{QuizRenderer-CLQqGwub.js → QuizRenderer-DaPQ7ySS.js} +1 -1
- package/web/.output/public/assets/{RenderingSmoke-DRHji-yz.js → RenderingSmoke-D3aD8prv.js} +1 -1
- package/web/.output/public/assets/{Select-BE2GfDud.js → Select-BYfJxqbI.js} +1 -1
- package/web/.output/public/assets/{SharedSession-DAM8ydVu.js → SharedSession-CEsTcr4j.js} +3 -3
- package/web/.output/public/assets/{ThemeToggle-BdGijxoN.js → ThemeToggle-BZH64Rpu.js} +1 -1
- package/web/.output/public/assets/{TrainingData-7EbVIzmQ.js → TrainingData-DujNU5sy.js} +3 -3
- package/web/.output/public/assets/TrajectoryReview-OjkaW6a6.js +35 -0
- package/web/.output/public/assets/TrajectoryReviewIndex-C5QAijKT.js +1 -0
- package/web/.output/public/assets/TrajectoryReviewWorkspace-DSn96Dxb.js +1 -0
- package/web/.output/public/assets/{Usage-BU1OfvzO.js → Usage-C5HU0Yc5.js} +1 -1
- package/web/.output/public/assets/activity-DdJDHeam.js +1 -0
- package/web/.output/public/assets/{anthropic-messages-rElikI6M.js → anthropic-messages-ZiHu59Q7.js} +1 -1
- package/web/.output/public/assets/{azure-openai-responses-yppPdVFy.js → azure-openai-responses-BHgsWKby.js} +1 -1
- package/web/.output/public/assets/calibration-7gf-WNR5.js +1 -0
- package/web/.output/public/assets/circle-question-mark-CK-DUGIA.js +1 -0
- package/web/.output/public/assets/clipboard-list-CRJvmiKL.js +1 -0
- package/web/.output/public/assets/{course-assembly-BIf5lrcx.js → course-assembly-CQE_9Vh4.js} +3 -3
- package/web/.output/public/assets/{course-submission-judgement-m2Gj1AKW.js → course-submission-judgement-Dmmw__tF.js} +1 -1
- package/web/.output/public/assets/{custom-tts-DapOpVQe.js → custom-tts-DqDZf3aF.js} +1 -1
- package/web/.output/public/assets/desktop-offline-IzFtwAhh.js +3 -0
- package/web/.output/public/assets/{dist-BudWH94z.js → dist-Dq5mGdi-.js} +1 -1
- package/web/.output/public/assets/file-text-CANyHShy.js +1 -0
- package/web/.output/public/assets/{from-study-plan-XEoNCZH-.js → from-study-plan-D14XtwuX.js} +2 -2
- package/web/.output/public/assets/{gemini-live-C2U-Xfzz.js → gemini-live-CGEtXm6j.js} +1 -1
- package/web/.output/public/assets/{google-generative-ai-8HY6m7rq.js → google-generative-ai-DcqlQJ6d.js} +1 -1
- package/web/.output/public/assets/{google-vertex-DnCA4qzU.js → google-vertex-C5RXjS_e.js} +1 -1
- package/web/.output/public/assets/{gpt-live-CEMTjMri.js → gpt-live-C4ltEjG_.js} +1 -1
- package/web/.output/public/assets/{index-C8C76R86.js → index-DCgpDQKf.js} +30 -30
- package/web/.output/public/assets/info-CGMF4Cdc.js +1 -0
- package/web/.output/public/assets/keating-stream-g73cFcJe.js +11 -0
- package/web/.output/public/assets/{learner-context-D0BfhTjc.js → learner-context-CpkZOwpe.js} +1 -1
- package/web/.output/public/assets/{library-pPmUSBEY.js → library-Dt_HrttW.js} +5 -5
- package/web/.output/public/assets/{live-context-CYuoXPUI.js → live-context-D5leiub7.js} +1 -1
- package/web/.output/public/assets/{mistral-conversations-BehK5A17.js → mistral-conversations-BSUNv6T_.js} +1 -1
- package/web/.output/public/assets/{openai-codex-responses-vVGnXFbc.js → openai-codex-responses-BEjWFOsh.js} +1 -1
- package/web/.output/public/assets/{openai-completions-SUIeYkEW.js → openai-completions-XWr91hin.js} +1 -1
- package/web/.output/public/assets/{openai-realtime-BSeGY5DQ.js → openai-realtime-BERRi9aO.js} +1 -1
- package/web/.output/public/assets/{openai-responses-DzgNIiYy.js → openai-responses-DeNWT5uB.js} +1 -1
- package/web/.output/public/assets/{openai-responses-shared-CHf6ek3_.js → openai-responses-shared-BdPh5feu.js} +1 -1
- package/web/.output/public/assets/{openai-tts-DRXdgxDT.js → openai-tts-ChGSz4r8.js} +1 -1
- package/web/.output/public/assets/{operation-BMdkNnIS.js → operation-kDIu8RLw.js} +2 -2
- package/web/.output/public/assets/{pi-messages-CpI94BCF.js → pi-messages-pzwJsgas.js} +1 -1
- package/web/.output/public/assets/{public-client-KRBnkgBr.js → public-client-CqivHtfG.js} +1 -1
- package/web/.output/public/assets/{recipes-BcSW8MlI.js → recipes-lg2hVa6Q.js} +1 -1
- package/web/.output/public/assets/{renderer-CKD_Hn-d.js → renderer-Cx1jCM2v.js} +1 -1
- package/web/.output/public/assets/{segments-nMdXcVhb.js → segments-BuQ4O8Bd.js} +1 -1
- package/web/.output/public/assets/{shared-actions-Bj56xHM3.js → shared-actions-B9eYiZAR.js} +1 -1
- package/web/.output/public/assets/{shared-renderer-C-yYfXvq.js → shared-renderer-hMxAoPZb.js} +7 -7
- package/web/.output/public/assets/shared-sessions-CwS3OLWR.js +5 -0
- package/web/.output/public/assets/{srs-WAeRnL8v.js → srs-BBTX4fCR.js} +1 -1
- package/web/.output/public/assets/{tavus-live-CVb9mE_J.js → tavus-live-HpBIH49i.js} +1 -1
- package/web/.output/public/assets/{topic-categorization-DHcL_IrS.js → topic-categorization-DfJaArmq.js} +1 -1
- package/web/.output/public/assets/trajectory-artifacts-EvNqJEHH.js +3 -0
- package/web/.output/public/assets/trajectory-store-C3qzwrr6.js +2 -0
- package/web/.output/public/assets/{transformers.web-CUMRYLoV.js → transformers.web-CrNvRfsr.js} +1 -1
- package/web/.output/public/assets/use-review-passes-DmWsvYdF.js +8 -0
- package/web/.output/public/assets/{use-sessions-C33USzrV.js → use-sessions-3UElISPk.js} +1 -1
- package/web/.output/public/assets/{useCoursesAccess-Zfb1Jqw-.js → useCoursesAccess-BgOspZex.js} +1 -1
- package/web/.output/public/assets/user-DEmM1U0O.js +1 -0
- package/web/.output/public/index.html +7 -7
- package/web/.output/public/sw.js +1 -1
- package/web/.output/server/_chunks/renderer-template.mjs +1 -1
- package/web/.output/server/_libs/h3+rou3+srvx.mjs +1 -1
- package/web/.output/server/index.mjs +1191 -1112
- package/web/.output/public/assets/CourseWorkspace-McgEEARK.js +0 -25
- package/web/.output/public/assets/ModelSelector-CiiiT7sM.js +0 -1
- package/web/.output/public/assets/OpenUIPreview-C8sgkYAj.js +0 -1
- package/web/.output/public/assets/PatternDigestPanel-Ctm7smvT.js +0 -1
- package/web/.output/public/assets/TrajectoryReview-DkC_NtyR.js +0 -35
- package/web/.output/public/assets/TrajectoryReviewIndex-C7pg9K9U.js +0 -1
- package/web/.output/public/assets/TrajectoryReviewWorkspace-D80QHrDn.js +0 -1
- package/web/.output/public/assets/calibration-B8NNSy6-.js +0 -1
- package/web/.output/public/assets/desktop-offline-BOKnmvYX.js +0 -2
- package/web/.output/public/assets/keating-stream-DeWgIdIv.js +0 -11
- package/web/.output/public/assets/shared-sessions-B2FsK7oM.js +0 -5
- package/web/.output/public/assets/trajectory-artifacts-B47xntcj.js +0 -3
- package/web/.output/public/assets/trajectory-store-CFat_se9.js +0 -2
- package/web/.output/public/assets/use-review-passes-BTc5nGD9.js +0 -8
package/web/.output/public/assets/{transformers.web-CUMRYLoV.js → transformers.web-CrNvRfsr.js}
RENAMED
|
@@ -36,4 +36,4 @@ ${r}${a}`+i.repeat(e)+`${r}`,o}function Fp(e,t,n,r){return`${t}${r}`+n.repeat(e)
|
|
|
36
36
|
`}var Jp=class extends K{static tokenizer_class=G;static image_processor_class=wp;static uses_processor_config=!1;async _call(e,t=null,n={}){t||=(N.warn(`You are using PaliGemma without a text prefix. It will perform as a picture-captioning model.`),``),Array.isArray(e)||(e=[e]),Array.isArray(t)||(t=[t]);let r=this.tokenizer.bos_token,i=this.image_processor.config.image_seq_length,a;t.some(e=>e.includes(Kp))?a=t.map(e=>{let t=e.replaceAll(Kp,Kp.repeat(i)),n=t.lastIndexOf(Kp),a=n===-1?0:n+Kp.length;return t.slice(0,a)+r+t.slice(a)+`
|
|
37
37
|
`}):(N.warn("You are passing both `text` and `images` to `PaliGemmaProcessor`. The processor expects special image tokens in the text, as many tokens as there are images per each text. It is recommended to add `<image>` tokens in the very beginning of your text. For this call, we will infer how many images each text has and add special tokens."),a=t.map(t=>qp(t,r,i,Kp,e.length)));let o=this.tokenizer(a,n);return{...await this.image_processor(e,n),...o}}},Yp=`<|image|>`,Xp=/<\|image_\d+\|>/g,Zp=class extends K{static image_processor_class=wp;static tokenizer_class=G;async _call(e,t=null,{padding:n=!0,truncation:r=!0,num_crops:i=null}={}){Array.isArray(e)||(e=[e]);let a,o;if(t){o=await this.image_processor(t,{num_crops:i});let{num_img_tokens:s}=o,c=e.map((e,t)=>e.split(Xp).join(Yp.repeat(s[t])));a=this.tokenizer(c,{padding:n,truncation:r});let l=this.tokenizer._tokenizer.token_to_id(Yp);a.input_ids.map_(e=>e==l?-e:e)}else a=this.tokenizer(e);return{...a,...o}}},Qp=class extends K{static tokenizer_class=G;static image_processor_class=wp;static uses_processor_config=!0;async _call(e,t=null,n={}){let r=await this.image_processor(e,n);if(t){let[e,n]=r.pixel_values.dims.slice(-2),{image_token:i,image_break_token:a,image_end_token:o,patch_size:s,spatial_merge_size:c}=this.config,l=s*c,u=Math.floor(e/l),d=Math.floor(n/l);t=structuredClone(t),Array.isArray(t)||(t=[t]);for(let e=0;e<t.length;++e){let n=i.repeat(d),r=n+a,s=n+o,c=r.repeat(u-1)+s;t[e]=t[e].replace(i,c)}}let i=t?this.tokenizer(t,n):{};return{...r,...i}}},$p=class extends K{static feature_extractor_class=Td;async _call(e){return await this.feature_extractor(e)}post_process_speaker_diarization(...e){return this.feature_extractor.post_process_speaker_diarization(...e)}get sampling_rate(){return this.feature_extractor.config.sampling_rate}},em=class extends kp{},tm=class extends em{},nm=class extends K{static image_processor_class=wp;async _call(...e){return await this.image_processor(...e)}post_process_masks(...e){return this.image_processor.post_process_masks(...e)}reshape_input_points(...e){return this.image_processor.reshape_input_points(...e)}},rm=class extends nm{},im=class extends rm{},am=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;async _call(e){return await this.feature_extractor(e)}},om=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;static uses_processor_config=!0;async _call(e,t=null,n={}){if(Array.isArray(e))throw Error(`Batched inputs are not supported yet.`);let r={};if(t){let i=t.length,{input_features:a}=await this.feature_extractor(t,{...n,max_length:i}),o=Math.round(i/this.config.encoder_ds_factor+1e-4),s=1+Math.ceil(o/this.config.stack_factor);r.audio_token_len=[s],r.audio_values=a;let c=this.config.audio_placeholder;if(!e.includes(c))throw Error(`The input text does not contain the image token ${c}.`);e=e.replaceAll(c,c.repeat(s))}return{...this.tokenizer(e,{add_special_tokens:!1,...n}),...r}}},sm=`[AUDIO]`,cm=`[BEGIN_AUDIO]`,lm=375;function um(e,t){let n=[];for(let r=0;r<e.length;r+=t)n.push(e.subarray(r,Math.min(r+t,e.length)));return n}var dm=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;static uses_processor_config=!1;async _call(e,t=null,n={}){if(Array.isArray(e))throw Error(`Batched inputs are not supported yet.`);let r={};if(t){if(!e.includes(sm))throw Error(`The input text does not contain the audio token ${sm}.`);Array.isArray(t)||(t=[t]);let i=e.split(sm),a=i.length-1;if(a!==t.length)throw Error(`The number of audio inputs (${t.length}) does not match the number of audio tokens in the text (${a}).`);let o=this.feature_extractor.config.n_samples,s=t.map(e=>um(e,o)),c=s.map(e=>e.length),l=s.flat(),u=(await Promise.all(l.map(e=>this.feature_extractor(e,n)))).map(e=>e.input_features);r.audio_values=u.length>1?fl(u,0):u[0];let d=i[0];for(let e=0;e<c.length;++e){d+=cm;for(let t=0;t<c[e];++t)d+=sm.repeat(lm);d+=i[e+1]}e=d}return{...this.tokenizer(e,{add_special_tokens:!1,...n}),...r}}},fm=32,pm=6,mm=8,hm=10,gm=32,_m=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;static uses_processor_config=!1;get num_mel_frames_first_audio_chunk(){return(pm+1)*mm}get num_samples_first_audio_chunk(){let{hop_length:e,n_fft:t}=this.feature_extractor.config;return(this.num_mel_frames_first_audio_chunk-1)*e+Math.floor(t/2)}get num_samples_per_audio_chunk(){let{hop_length:e,n_fft:t}=this.feature_extractor.config;return mm*e+t}get num_right_pad_tokens(){return pm+1+hm}get audio_length_per_tok(){return mm}get raw_audio_length_per_tok(){return mm*this.feature_extractor.config.hop_length}async _call(e,{is_streaming:t=!1,is_first_audio_chunk:n=!0}={}){if(Hu(e,`VoxtralRealtimeProcessor`),!t&&!n)throw Error("In non-streaming mode (`is_streaming=false`), `is_first_audio_chunk` must be `true`.");if(n)if(t){let t=fm*this.raw_audio_length_per_tok,n=new Float32Array(t+e.length);n.set(e,t);let r=await this.feature_extractor(n,{center:!0}),i=1+(fm+pm),a=new BigInt64Array(i).fill(BigInt(gm));return a[0]=1n,{input_ids:new U(`int64`,a,[1,i]),...r}}else{let t=this.num_right_pad_tokens*this.raw_audio_length_per_tok,n=new Float32Array(e.length+t);return n.set(e),await this.feature_extractor(n,{center:!0})}else return await this.feature_extractor(e,{center:!1})}},vm=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;async _call(e){return await this.feature_extractor(e)}},ym=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;async _call(e){return await this.feature_extractor(e)}},bm=class extends K{static tokenizer_class=G;static feature_extractor_class=Nd;async _call(e){return await this.feature_extractor(e)}},xm=class{static async from_pretrained(e,t={}){let n=await rc(e,Lu,!0,t),{image_processor_type:r,feature_extractor_type:i,processor_class:a}=n;if(a&&Bu[a])return Bu[a].from_pretrained(e,t);if(!r&&!i)throw Error("No `image_processor_type` or `feature_extractor_type` found in the config.");let o={};if(r){let e=tf[r.replace(/Fast$/,``)];if(!e)throw Error(`Unknown image_processor_type: '${r}'.`);o.image_processor=new e(n)}if(i){let e=tf[i];if(e)o.image_processor=new e(n);else{let e=Uu[i];if(!e)throw Error(`Unknown feature_extractor_type: '${i}'.`);o.feature_extractor=new e(n)}}return new K({},o,null)}};async function Sm(e,t){return await rc(e,`config.json`,!0,t)}function Cm(e){let t={},n={};switch(e.model_type){case`llava`:case`paligemma`:case`gemma3`:case`florence2`:case`llava_onevision`:case`idefics3`:case`granite_speech`:case`ultravox`:case`voxtral`:case`voxtral_realtime`:case`smolvlm`:case`gemma3n`:case`gemma4`:case`lfm2_vl`:case`chatterbox`:case`lighton_ocr`:case`glm_ocr`:case`mistral3`:case`qwen2_5_vl`:case`qwen3_vl`:case`qwen3_vl_moe`:n=Cm(e.text_config);break;case`moondream1`:n=Cm(e.phi_config);break;case`musicgen`:n=Cm(e.decoder);break;case`multi_modality`:n=Cm(e.language_config);break;case`gpt2`:case`gptj`:case`jais`:case`codegen`:case`gpt_bigcode`:t.num_heads=`n_head`,t.num_layers=`n_layer`,t.hidden_size=`n_embd`;break;case`gpt_neox`:case`stablelm`:case`opt`:case`falcon`:case`modernbert-decoder`:t.num_heads=`num_attention_heads`,t.num_layers=`num_hidden_layers`,t.hidden_size=`hidden_size`;break;case`gpt_oss`:case`llama`:case`llama4_text`:case`nanochat`:case`apertus`:case`arcee`:case`afmoe`:case`lfm2`:case`lfm2_moe`:case`smollm3`:case`olmo`:case`olmo2`:case`olmo3`:case`mobilellm`:case`granite`:case`granitemoehybrid`:case`cohere`:case`cohere2`:case`mistral`:case`voxtral_realtime_text`:case`voxtral_realtime_encoder`:case`starcoder2`:case`qwen2`:case`qwen2_moe`:case`qwen2_vl`:case`qwen2_vl_text`:case`qwen2_5_vl_text`:case`qwen3_moe`:case`qwen3_vl_text`:case`qwen3_vl_moe_text`:case`phi`:case`phi3`:case`phi3_v`:case`llava_qwen2`:t.num_heads=`num_key_value_heads`,t.num_layers=`num_hidden_layers`,t.hidden_size=`hidden_size`,t.num_attention_heads=`num_attention_heads`,t.dim_kv=`head_dim`;break;case`qwen3`:case`solar_open`:case`glm_ocr_text`:case`gemma`:case`gemma2`:case`vaultgemma`:case`gemma3_text`:case`gemma3n_text`:case`gemma4_text`:case`glm`:case`helium`:case`ernie4_5`:case`hunyuan_v1_dense`:case`falcon_h1`:case`nemotron_h`:case`ministral`:case`ministral3`:t.num_heads=`num_key_value_heads`,t.num_layers=`num_hidden_layers`,t.dim_kv=`head_dim`;break;case`openelm`:t.num_heads=`num_kv_heads`,t.num_layers=`num_transformer_layers`,t.dim_kv=`head_dim`;break;case`gpt_neo`:case`donut-swin`:t.num_heads=`num_heads`,t.num_layers=`num_layers`,t.hidden_size=`hidden_size`;break;case`bloom`:t.num_heads=`n_head`,t.num_layers=`n_layer`,t.hidden_size=`hidden_size`;break;case`mpt`:t.num_heads=`n_heads`,t.num_layers=`n_layers`,t.hidden_size=`d_model`;break;case`exaone`:t.num_heads=`num_key_value_heads`,t.num_layers=`num_layers`,t.dim_kv=`head_dim`,t.num_attention_heads=`num_attention_heads`;break;case`youtu`:case`deepseek_v3`:case`glm_moe_dsa`:case`mistral4`:t.num_heads=`num_key_value_heads`,t.num_layers=`num_hidden_layers`,t.dim_kv=`qk_head_dim`,t.num_attention_heads=`num_attention_heads`;break;case`t5`:case`mt5`:case`longt5`:t.num_decoder_layers=`num_decoder_layers`,t.num_decoder_heads=`num_heads`,t.decoder_dim_kv=`d_kv`,t.num_encoder_layers=`num_layers`,t.num_encoder_heads=`num_heads`,t.encoder_dim_kv=`d_kv`;break;case`bart`:case`mbart`:case`marian`:case`whisper`:case`lite-whisper`:case`m2m_100`:case`blenderbot`:case`blenderbot-small`:case`florence2_language`:t.num_decoder_layers=`decoder_layers`,t.num_decoder_heads=`decoder_attention_heads`,t.decoder_hidden_size=`d_model`,t.num_encoder_layers=`encoder_layers`,t.num_encoder_heads=`encoder_attention_heads`,t.encoder_hidden_size=`d_model`;break;case`speecht5`:t.num_decoder_layers=`decoder_layers`,t.num_decoder_heads=`decoder_attention_heads`,t.decoder_hidden_size=`hidden_size`,t.num_encoder_layers=`encoder_layers`,t.num_encoder_heads=`encoder_attention_heads`,t.encoder_hidden_size=`hidden_size`;break;case`trocr`:t.num_encoder_layers=t.num_decoder_layers=`decoder_layers`,t.num_encoder_heads=t.num_decoder_heads=`decoder_attention_heads`,t.encoder_hidden_size=t.decoder_hidden_size=`d_model`;break;case`musicgen_decoder`:t.num_encoder_layers=t.num_decoder_layers=`num_hidden_layers`,t.num_encoder_heads=t.num_decoder_heads=`num_attention_heads`,t.encoder_hidden_size=t.decoder_hidden_size=`hidden_size`;break;case`moonshine`:t.num_decoder_layers=`decoder_num_hidden_layers`,t.num_decoder_heads=`decoder_num_key_value_heads`,t.num_encoder_layers=`encoder_num_hidden_layers`,t.num_encoder_heads=`encoder_num_key_value_heads`,t.encoder_hidden_size=t.decoder_hidden_size=`hidden_size`;break;case`cohere_asr`:t.num_decoder_layers=`num_hidden_layers`,t.num_decoder_heads=`num_key_value_heads`,t.decoder_hidden_size=`hidden_size`,t.decoder_dim_kv=`head_dim`;let{num_hidden_layers:r,num_attention_heads:i,hidden_size:a}=e.encoder_config;n={num_encoder_layers:r,num_encoder_heads:i,encoder_hidden_size:a,encoder_dim_kv:e.head_dim};break;case`vision-encoder-decoder`:let o=Cm(e.decoder),s=`num_decoder_layers`in o,c=di(e,[`model_type`,`is_encoder_decoder`]);return s?(c.num_decoder_layers=o.num_decoder_layers,c.num_decoder_heads=o.num_decoder_heads,c.decoder_hidden_size=o.decoder_hidden_size,c.num_encoder_layers=o.num_encoder_layers,c.num_encoder_heads=o.num_encoder_heads,c.encoder_hidden_size=o.encoder_hidden_size):(c.num_layers=o.num_layers,c.num_heads=o.num_heads,c.hidden_size=o.hidden_size),c}let r={...n,...di(e,[`model_type`,`multi_query`,`is_encoder_decoder`])};for(let n in t)r[n]=e[t[n]];return r}function wm(e,t){e instanceof Em||(e=new Em(e));let n=t?.prefix??`past_key_values`,r=n===`present`?`present`:`past`,i=new Set;if([`lfm2`,`lfm2_moe`].includes(e.model_type)){let{layer_types:t}=e;for(let e=0;e<t.length;++e)if(t[e]===`full_attention`)i.add(`${n}.${e}.key`),i.add(`${n}.${e}.value`);else if(t[e]===`conv`)i.add(`${r}_conv.${e}`);else throw Error(`Unsupported layer type: ${t[e]}`);return i}else if([`granitemoehybrid`,`falcon_h1`,`nemotron_h`].includes(e.model_type)){let t=e,a=t.layer_types??t.layers_block_type,o=t.num_hidden_layers??a?.length;for(let e=0;e<o;++e)(!a||a[e]===`mamba`)&&(i.add(`${r}_conv.${e}`),i.add(`${r}_ssm.${e}`)),(!a||a[e]===`attention`)&&(i.add(`${n}.${e}.key`),i.add(`${n}.${e}.value`));return i}else if([`qwen3_next`,`qwen3_5_text`,`qwen3_5_moe_text`,`olmo_hybrid`].includes(e.model_type)){let{layer_types:t}=e;for(let a=0;a<t.length;++a)if(t[a]===`full_attention`)i.add(`${n}.${a}.key`),i.add(`${n}.${a}.value`);else if(t[a]===`linear_attention`)e.model_type===`olmo_hybrid`?(i.add(`${r}_conv.${a}.key`),i.add(`${r}_conv.${a}.value`),i.add(`${r}_conv.${a}.query`)):i.add(`${r}_conv.${a}`),i.add(`${r}_recurrent.${a}`);else throw Error(`Unsupported layer type: ${t[a]}`);return i}else if([`gemma4`,`gemma4_text`].includes(e.model_type)){let t=e.model_type===`gemma4`?e.text_config:e,r=t.num_hidden_layers-(t.num_kv_shared_layers??0);for(let e=0;e<r;++e)i.add(`${n}.${e}.key`),i.add(`${n}.${e}.value`);return i}else if([`lfm2_vl`,`qwen3_5`,`qwen3_5_moe`,`voxtral_realtime`].includes(e.model_type)){let n;return n=e.model_type===`voxtral_realtime`&&t?.session_name===`audio_encoder`?e.audio_config:e.text_config,wm(n,t)}return Tm(e,{prefix:n})}function Tm(e,{prefix:t=`past_key_values`}={}){let n=new Set,r=e.normalized_config;if(r.is_encoder_decoder&&`num_encoder_heads`in r&&`num_decoder_heads`in r)for(let e=0;e<r.num_decoder_layers;++e)n.add(`${t}.${e}.encoder.key`),n.add(`${t}.${e}.encoder.value`),n.add(`${t}.${e}.decoder.key`),n.add(`${t}.${e}.decoder.value`);else if(r.multi_query)for(let e=0;e<r.num_layers;++e)n.add(`${t}.${e}.key_value`);else for(let e=0;e<r.num_layers;++e)n.add(`${t}.${e}.key`),n.add(`${t}.${e}.value`);return n}var Em=class e{model_type=null;is_encoder_decoder=!1;max_position_embeddings;"transformers.js_config";constructor(e){Object.assign(this,e),this.normalized_config=Cm(this)}static async from_pretrained(t,{progress_callback:n=null,config:r=null,cache_dir:i=null,local_files_only:a=!1,revision:o=`main`}={}){r&&!(r instanceof e)&&(r=new e(r));let s=r??await Sm(t,{progress_callback:n,config:r,cache_dir:i,local_files_only:a,revision:o});return new this(s)}},Dm=class{static async from_pretrained(...e){return Em.from_pretrained(...e)}};function Om(e,t,n){return e?typeof e==`object`&&e?e.hasOwnProperty(t)?+e[t]:e.hasOwnProperty(n)?+e[n]:0:+e:0}function km(e,t){let n=[];for(let r=0;r<t;++r)n.push(`${e}_data${r===0?``:`_`+r}`);return n}async function Am(e,t,n,r){let i=`${t}${r}.onnx`;return await tc(e,`${n.subfolder??``}/${i}`,!0,n,j.IS_NODE_ENV)}async function jm(e,t,n,r,i,a={}){let o=`${t}${n}.onnx`,s=j.IS_NODE_ENV,c=[],l=Om(i,o,t);if(l>0){if(l>Os)throw Error(`The number of external data chunks (${l}) exceeds the maximum allowed value (${Os}).`);let t=km(o,l);for(let n of t){let t=`${r.subfolder??``}/${n}`;c.push(new Promise(async(i,a)=>{let o=await tc(e,t,!0,r,s);i(o instanceof Uint8Array?{path:n,data:o}:n)}))}}else a.externalData!==void 0&&(c=a.externalData.map(async t=>{if(typeof t.data==`string`){let n=await tc(e,t.data,!0,r);return{...t,data:n}}return t}));return Promise.all(c)}async function Mm(e,t,n,r=!1,i=void 0){let a=n.config?.[`transformers.js_config`]??{},o=Kc(n.device??a.device,t,{warn:e=>N.info(e)}),s=Mc(o),c=a.device_config??{};c.hasOwnProperty(o)&&(a={...a,...c[o]});let l=Qc(n.dtype??a.dtype,t,o,{configDtype:a.dtype,warn:e=>N.info(e)});if(!Zc.hasOwnProperty(l))throw Error(`Invalid dtype: ${l}. Should be one of: ${Object.keys(Jc).join(`, `)}`);if(o===`webgpu`&&!j.IS_NODE_ENV&&l===Jc.fp16&&!await qc())throw Error(`The device (${o}) does not support fp16.`);let u=Zc[l],d={...n.session_options};d.executionProviders??=s;let f=a.free_dimension_overrides;f?d.freeDimensionOverrides??=f:o.startsWith(`webnn`)&&!d.freeDimensionOverrides&&N.warn(`WebNN does not currently support dynamic shapes and requires 'free_dimension_overrides' to be set in config.json, preferably as a field within config["transformers.js_config"]["device_config"]["${o}"]. When 'free_dimension_overrides' is not set, you may experience significant performance degradation.`);let p=Am(e,t,n,u),m=await jm(e,t,u,n,n.use_external_data_format??a.use_external_data_format,d);if(m.length>0&&(!j.IS_NODE_ENV||m.some(e=>typeof e!=`string`))&&(d.externalData=m),r&&o===`webgpu`){let e=wm(n.config,{prefix:`present`,session_name:i});if(e.size>0&&!Vc()){let t={};for(let n of e)t[n]=`gpu-buffer`;d.preferredOutputLocation=t}}return{buffer_or_path:await p,session_options:d,session_config:{dtype:l,device:o}}}async function Nm(e,t,n,r=void 0){return Object.fromEntries(await Promise.all(Object.keys(t).map(async i=>{let a=r?.[i]??!1,{buffer_or_path:o,session_options:s,session_config:c}=await Mm(e,t[i],n,a,i);return[i,await Ic(o,s,c)]})))}function Pm(e){for(let t in e)zc(e[t])?e[t]=new U(e[t]):typeof e[t]==`object`&&Pm(e[t]);return e}async function J(e,t){let n=Fm(e,t);try{return Pm(await Rc(e,Object.fromEntries(Object.entries(n).map(([e,t])=>{let n=t.ort_tensor;return j.IS_NODE_ENV&&typeof Float16Array<`u`&&n.cpuData instanceof Float16Array&&(n.cpuData=new Uint16Array(n.cpuData.buffer)),[e,n]}))))}catch(e){let t=Object.fromEntries(Object.entries(n).map(([e,t])=>{let n={type:t.type,dims:t.dims,location:t.location};return n.location!==`gpu-buffer`&&(n.data=t.data),[e,n]}));throw N.error(`An error occurred during model execution: "${e}".`),N.error(`Inputs given to model:`,t),e}}function Fm(e,t){let n=Object.create(null),r=[];for(let i of e.inputNames){let e=t[i];if(!(e instanceof U)){r.push(i);continue}n[i]=Vc()?e.clone():e}if(r.length>0)throw Error(`An error occurred during model execution: "Missing the following inputs: ${r.join(`, `)}.`);let i=Object.keys(t).length,a=e.inputNames.length;if(i>a){let n=Object.keys(t).filter(t=>!e.inputNames.includes(t));N.warn(`WARNING: Too many inputs were provided (${i} > ${a}). The following inputs will be ignored: "${n.join(`, `)}".`)}return n}var Im=class{},Y=class extends Im{constructor({logits:e,...t}){super(),this.logits=e;let n=Object.values(t);n.length>0&&(this.attentions=n)}},Lm=class extends Im{constructor({logits:e}){super(),this.logits=e}},Rm=class extends Im{constructor({logits:e}){super(),this.logits=e}},zm=class extends Im{constructor({start_logits:e,end_logits:t}){super(),this.start_logits=e,this.end_logits=t}},Bm=class extends Im{constructor({logits:e}){super(),this.logits=e}},Vm=class extends Im{constructor({alphas:e}){super(),this.alphas=e}},Hm=class extends ni{_call(e,t){throw Error("`_call` should be implemented in a subclass")}},Um=class extends ni{_call(e,t){throw Error("`_call` should be implemented in a subclass")}},Wm=class extends ni{constructor(){super(),this.processors=[]}push(e){this.processors.push(e)}extend(e){this.processors.push(...e)}_call(e,t){let n=t;for(let t of this.processors)n=t(e,n);return n}[Symbol.iterator](){return this.processors.values()}},Gm=class extends Hm{constructor(e){super(),this.bos_token_id=e}_call(e,t){for(let n=0;n<e.length;++n)if(e[n].length===1){let e=t[n].data;e.fill(-1/0),e[this.bos_token_id]=0}return t}},Km=class extends Hm{constructor(e,t){super(),this.max_length=e,this.eos_token_id=Array.isArray(t)?t:[t]}_call(e,t){for(let n=0;n<e.length;++n)if(e[n].length===this.max_length-1){let e=t[n].data;e.fill(-1/0);for(let t of this.eos_token_id)e[t]=0}return t}},qm=class extends Hm{constructor(e){super(),this.suppress_tokens=e}_call(e,t){for(let n=0;n<e.length;++n){let e=t[n].data;for(let t of this.suppress_tokens)e[t]=-1/0}return t}},Jm=class extends Hm{constructor(e,t){super(),this.begin_suppress_tokens=e,this.begin_index=t}_call(e,t){for(let n=0;n<e.length;++n)if(e[n].length===this.begin_index){let e=t[n].data;for(let t of this.begin_suppress_tokens)e[t]=-1/0}return t}},Ym=class extends Hm{constructor(e,t){super(),this.eos_token_id=Array.isArray(e.eos_token_id)?e.eos_token_id[0]:e.eos_token_id,this.no_timestamps_token_id=e.no_timestamps_token_id,this.timestamp_begin=this.no_timestamps_token_id+1,this.begin_index=t.length,t.at(-1)===this.no_timestamps_token_id&&--this.begin_index,this.max_initial_timestamp_index=e.max_initial_timestamp_index}_call(e,t){for(let n=0;n<e.length;++n){let r=t[n].data;if(r[this.no_timestamps_token_id]=-1/0,e[n].length===this.begin_index){r.subarray(0,this.timestamp_begin).fill(-1/0);continue}let i=e[n].slice(this.begin_index),a=i.length>=1&&i[i.length-1]>=this.timestamp_begin,o=i.length<2||i[i.length-2]>=this.timestamp_begin;if(a&&(o?r.subarray(this.timestamp_begin).fill(-1/0):r.subarray(0,this.eos_token_id).fill(-1/0)),e[n].length===this.begin_index&&this.max_initial_timestamp_index!==null){let e=this.timestamp_begin+this.max_initial_timestamp_index;r.subarray(e+1).fill(-1/0)}let s=sc(r);Math.log(s.subarray(this.timestamp_begin).map(Math.exp).reduce((e,t)=>e+t))>lc(s.subarray(0,this.timestamp_begin))[0]&&r.subarray(0,this.timestamp_begin).fill(-1/0)}return t}},Xm=class extends Hm{constructor(e){super(),this.no_repeat_ngram_size=e}getNgrams(e){let t=e.length,n=[];for(let r=0;r<t+1-this.no_repeat_ngram_size;++r){let t=[];for(let n=0;n<this.no_repeat_ngram_size;++n)t.push(e[r+n]);n.push(t.map(Number))}let r=new Map;for(let e of n){let t=e.slice(0,e.length-1),n=JSON.stringify(t),i=r.get(n)??[];i.push(e[e.length-1]),r.set(n,i)}return r}getGeneratedNgrams(e,t){let n=t.slice(t.length+1-this.no_repeat_ngram_size,t.length);return e.get(JSON.stringify(n.map(Number)))??[]}calcBannedNgramTokens(e){let t=[];if(e.length+1<this.no_repeat_ngram_size)return t;{let t=this.getNgrams(e);return this.getGeneratedNgrams(t,e)}}_call(e,t){for(let n=0;n<e.length;++n){let r=t[n].data,i=this.calcBannedNgramTokens(e[n]);for(let e of i)r[e]=-1/0}return t}},Zm=class extends Hm{constructor(e){super(),this.penalty=e}_call(e,t){for(let n=0;n<e.length;++n){let r=t[n].data;for(let t of new Set(e[n])){let e=Number(t);r[e]<0?r[e]*=this.penalty:r[e]/=this.penalty}}return t}},Qm=class extends Hm{constructor(e,t){super(),this.min_length=e,this.eos_token_id=Array.isArray(t)?t:[t]}_call(e,t){for(let n=0;n<e.length;++n)if(e[n].length<this.min_length){let e=t[n].data;for(let t of this.eos_token_id)e[t]=-1/0}return t}},$m=class extends Hm{constructor(e,t,n){super(),this.prompt_length_to_skip=e,this.min_new_tokens=t,this.eos_token_id=Array.isArray(n)?n:[n]}_call(e,t){for(let n=0;n<e.length;++n)if(e[n].length-this.prompt_length_to_skip<this.min_new_tokens){let e=t[n].data;for(let t of this.eos_token_id)e[t]=-1/0}return t}},eh=class extends Hm{constructor(e,t){super(),this.bad_words_ids=e,this.eos_token_id=Array.isArray(t)?t:[t]}_call(e,t){for(let n=0;n<e.length;++n){let r=t[n].data,i=e[n];for(let e of this.bad_words_ids){if(i.length<e.length-1)continue;let t=!0;for(let n=1;n<=e.length-1;++n)if(e.at(-n-1)!=i.at(-n)){t=!1;break}t&&(r[e.at(-1)]=-1/0)}}return t}},th=class extends Hm{constructor(e){if(super(),e<=1)throw Error(`Require guidance scale >1 to use the classifier free guidance processor, got guidance scale ${e}.`);this.guidance_scale=e}_call(e,t){if(t.dims[0]!==2*e.length)throw Error(`Logits should have twice the batch size of the input ids, the first half of batches corresponding to the conditional inputs, and the second half of batches corresponding to the unconditional inputs. Got batch size ${t.dims[0]} for the logits and ${e.length} for the input ids.`);let n=e.length,r=t.slice([0,n],null),i=t.slice([n,t.dims[0]],null);for(let e=0;e<i.data.length;++e)i.data[e]+=(r.data[e]-i.data[e])*this.guidance_scale;return i}},nh=class extends Um{constructor(e){if(super(),typeof e!=`number`||e<=0){let t=`\`temperature\` (=${e}) must be a strictly positive float, otherwise your next token scores will be invalid.`;e===0&&(t+=" If you're looking for greedy decoding strategies, set `do_sample=false`.")}this.temperature=e}_call(e,t){let n=t.data;for(let e=0;e<n.length;++e)n[e]/=this.temperature;return t}},rh=class{max_length=20;max_new_tokens=null;min_length=0;min_new_tokens=null;early_stopping=!1;max_time=null;do_sample=!1;num_beams=1;num_beam_groups=1;penalty_alpha=null;use_cache=!0;temperature=1;top_k=50;top_p=1;typical_p=1;epsilon_cutoff=0;eta_cutoff=0;diversity_penalty=0;repetition_penalty=1;encoder_repetition_penalty=1;length_penalty=1;no_repeat_ngram_size=0;bad_words_ids=null;force_words_ids=null;renormalize_logits=!1;constraints=null;forced_bos_token_id=null;forced_eos_token_id=null;remove_invalid_values=!1;exponential_decay_length_penalty=null;suppress_tokens=null;streamer=null;begin_suppress_tokens=null;forced_decoder_ids=null;guidance_scale=null;num_return_sequences=1;output_attentions=!1;output_hidden_states=!1;output_scores=!1;return_dict_in_generate=!1;pad_token_id=null;bos_token_id=null;eos_token_id=null;encoder_no_repeat_ngram_size=0;decoder_start_token_id=null;generation_kwargs={};constructor(e){Object.assign(this,di(e,Object.getOwnPropertyNames(this)))}},ih=class extends ni{_call(e,t){throw Error(`StoppingCriteria needs to be subclassed`)}},ah=class e extends ni{constructor(){super(),this.criteria=[]}push(e){this.criteria.push(e)}extend(t){t instanceof e?t=t.criteria:t instanceof ih&&(t=[t]),this.criteria.push(...t)}_call(e,t){let n=Array(e.length).fill(!1);for(let r of this.criteria){let i=r(e,t);for(let e=0;e<n.length;++e)n[e]||=i[e]}return n}[Symbol.iterator](){return this.criteria.values()}},oh=class extends ih{constructor(e,t=null){super(),this.max_length=e,this.max_position_embeddings=t}_call(e){return e.map(e=>e.length>=this.max_length)}},sh=class extends ih{constructor(e){super(),Array.isArray(e)||(e=[e]),this.eos_token_id=e}_call(e,t){return e.map(e=>{let t=e.at(-1);return this.eos_token_id.some(e=>t==e)})}},ch=class extends ni{constructor(e){super(),this.generation_config=e}async _call(e){return this.sample(e)}async sample(e){throw Error(`sample should be implemented in subclasses.`)}getLogits(e,t){let n=e.dims.at(-1),r=e.data;if(t===-1)r=r.slice(-n);else{let e=t*n;r=r.slice(e,e+n)}return r}randomSelect(e){return ws(e)}static getSampler(e){if(e.do_sample)return new uh(e);if(e.num_beams>1)return new dh(e);if(e.num_return_sequences>1)throw Error(`num_return_sequences has to be 1 when doing greedy search, but is ${e.num_return_sequences}.`);return new lh(e)}},lh=class extends ch{async sample(e){let t=lc(e.data)[1];return[[BigInt(t),0]]}},uh=class extends ch{async sample(e){let t=e.dims.at(-1);this.generation_config.top_k>0&&(t=Math.min(this.generation_config.top_k,t));let[n,r]=await al(e,t),i=oc(n.data);return Array.from({length:this.generation_config.num_beams},()=>{let e=this.randomSelect(i);return[r.data[e],Math.log(i[e])]})}},dh=class extends ch{async sample(e){let t=e.dims.at(-1);this.generation_config.top_k>0&&(t=Math.min(this.generation_config.top_k,t));let[n,r]=await al(e,t),i=oc(n.data);return Array.from({length:this.generation_config.num_beams},(e,t)=>[r.data[t],Math.log(i[t])])}},fh=class{constructor(e){if(e)for(let t in e){if(t in this)throw TypeError(`Key "${t}" conflicts with an existing property on DynamicCache`);let n=e[t];if(!(n instanceof U))throw TypeError(`Expected a Tensor for key "${t}", got ${typeof n}`);this[t]=n}}get_seq_length(){let e=this;if(Object.keys(e).length===0)return 0;for(let t in e)if(t.startsWith(`past_key_values.`))return e[t].dims.at(-2);throw Error(`Unable to determine sequence length from the cache.`)}update(e){for(let t in e){let n=this[t],r=e[t];n&&n!==r&&n.location===`gpu-buffer`&&n.dispose(),this[t]=r}}async dispose(){let e=[];for(let t of Object.values(this))t.location===`gpu-buffer`&&e.push(t.dispose());await Promise.all(e)}},X={EncoderOnly:0,EncoderDecoder:1,Seq2Seq:2,Vision2Seq:3,DecoderOnly:4,DecoderOnlyWithoutHead:5,MaskGeneration:6,ImageTextToText:7,Musicgen:8,MultiModality:9,Phi3V:10,AudioTextToText:11,AutoEncoder:12,ImageAudioTextToText:13,Supertonic:14,Chatterbox:15,VoxtralRealtime:16},ph={[X.DecoderOnly]:{sessions:(e,t)=>({model:t.model_file_name??`model`}),cache_sessions:{model:!0},optional_configs:{generation_config:`generation_config.json`}},[X.DecoderOnlyWithoutHead]:{sessions:(e,t)=>({model:t.model_file_name??`model`})},[X.Seq2Seq]:{sessions:()=>({model:`encoder_model`,decoder_model_merged:`decoder_model_merged`}),cache_sessions:{decoder_model_merged:!0},optional_configs:{generation_config:`generation_config.json`}},[X.Vision2Seq]:{sessions:()=>({model:`encoder_model`,decoder_model_merged:`decoder_model_merged`}),cache_sessions:{decoder_model_merged:!0},optional_configs:{generation_config:`generation_config.json`}},[X.Musicgen]:{sessions:()=>({model:`text_encoder`,decoder_model_merged:`decoder_model_merged`,encodec_decode:`encodec_decode`}),cache_sessions:{decoder_model_merged:!0},optional_configs:{generation_config:`generation_config.json`}},[X.EncoderDecoder]:{sessions:()=>({model:`encoder_model`,decoder_model_merged:`decoder_model_merged`}),cache_sessions:{decoder_model_merged:!0}},[X.MaskGeneration]:{sessions:()=>({model:`vision_encoder`,prompt_encoder_mask_decoder:`prompt_encoder_mask_decoder`})},[X.ImageTextToText]:{text_only_sessions:{embed_tokens:`embed_tokens`,decoder_model_merged:`decoder_model_merged`},sessions:(e,t,n)=>{let r={...ph[X.ImageTextToText].text_only_sessions};return n||(r.vision_encoder=`vision_encoder`),e.is_encoder_decoder&&(r.model=`encoder_model`),r},cache_sessions:{decoder_model_merged:!0},optional_configs:{generation_config:`generation_config.json`}},[X.AudioTextToText]:{text_only_sessions:{embed_tokens:`embed_tokens`,decoder_model_merged:`decoder_model_merged`},sessions:(e,t,n)=>{let r={...ph[X.AudioTextToText].text_only_sessions};return n||(r.audio_encoder=`audio_encoder`),r},cache_sessions:{decoder_model_merged:!0},optional_configs:{generation_config:`generation_config.json`}},[X.ImageAudioTextToText]:{text_only_sessions:{embed_tokens:`embed_tokens`,decoder_model_merged:`decoder_model_merged`},sessions:(e,t,n)=>{let r={...ph[X.ImageAudioTextToText].text_only_sessions};return n||(r.audio_encoder=`audio_encoder`,r.vision_encoder=`vision_encoder`),r},optional_configs:{generation_config:`generation_config.json`}},[X.Phi3V]:{sessions:()=>({prepare_inputs_embeds:`prepare_inputs_embeds`,model:`model`,vision_encoder:`vision_encoder`}),cache_sessions:{model:!0},optional_configs:{generation_config:`generation_config.json`}},[X.MultiModality]:{sessions:()=>({prepare_inputs_embeds:`prepare_inputs_embeds`,model:`language_model`,lm_head:`lm_head`,gen_head:`gen_head`,gen_img_embeds:`gen_img_embeds`,image_decode:`image_decode`}),cache_sessions:{model:!0},optional_configs:{generation_config:`generation_config.json`}},[X.AutoEncoder]:{sessions:()=>({encoder_model:`encoder_model`,decoder_model:`decoder_model`})},[X.Supertonic]:{sessions:()=>({text_encoder:`text_encoder`,latent_denoiser:`latent_denoiser`,voice_decoder:`voice_decoder`})},[X.Chatterbox]:{sessions:()=>({embed_tokens:`embed_tokens`,speech_encoder:`speech_encoder`,model:`language_model`,conditional_decoder:`conditional_decoder`}),cache_sessions:{model:!0},optional_configs:{generation_config:`generation_config.json`}},[X.VoxtralRealtime]:{text_only_sessions:{embed_tokens:`embed_tokens`,decoder_model_merged:`decoder_model_merged`},sessions:(e,t,n)=>{let r={...ph[X.VoxtralRealtime].text_only_sessions};return n||(r.audio_encoder=`audio_encoder`),r},cache_sessions:{decoder_model_merged:!0,audio_encoder:!0},optional_configs:{generation_config:`generation_config.json`}},default:{sessions:(e,t)=>({model:t.model_file_name??`model`})}};function mh(e,t,n={}){let r=ph[e]??ph.default;return{sessions:r.sessions(t,n,n.textOnly??!1),cache_sessions:r.cache_sessions,optional_configs:r.optional_configs}}function hh(e,{warn:t=!0}={}){let n=e.architectures||[];for(let e of n){let t=wh.get(e);if(t!==void 0)return t}if(e.model_type){let t=wh.get(e.model_type);if(t!==void 0)return t;for(let t of Object.values(vh))if(t.has(e.model_type)){let n=wh.get(t.get(e.model_type));if(n!==void 0)return n}}if(t){let t=n.length>0?n.join(`, `):`(none)`;N.warn(`[resolve_model_type] Architecture(s) not found in MODEL_TYPE_MAPPING: [${t}] for model type '${e.model_type}'. Falling back to EncoderOnly (single model.onnx file). If you encounter issues, please report at: ${Fu}`)}return X.EncoderOnly}function gh(e,{config:t=null,cache_dir:n=null,local_files_only:r=!1,revision:i=`main`}={}){return t===null?Ws(JSON.stringify([e,n,r,i]),()=>Dm.from_pretrained(e,{config:t,cache_dir:n,local_files_only:r,revision:i})):Dm.from_pretrained(e,{config:t,cache_dir:n,local_files_only:r,revision:i})}async function _h(e,{config:t=null,dtype:n=null,device:r=null,model_file_name:i=null}={}){t=await gh(e,{config:t});let a=[`config.json`],o=t[`transformers.js_config`]??{},s=o.use_external_data_format,c=`onnx`,l=r??o.device,u=n??o.dtype,d=hh(t),f=(e,t=null)=>{t??=e;let n=Zc[Qc(u,e,Kc(l,e))]??``,r=`${t}${n}.onnx`,i=`${c}/${r}`;a.push(i);let o=Om(s,r,e);for(let e of km(r,o)){let t=`${c}/${e}`;a.push(t)}},{sessions:p,optional_configs:m}=mh(d,t,{model_file_name:i});for(let[e,t]of Object.entries(p))f(e,t);if(m)for(let e of Object.values(m))a.push(e);return a}var vh=null;function yh(e){vh=e}function bh(e){if(e instanceof U)return e;if(e.length===0)throw Error(`items must be non-empty`);if(Array.isArray(e[0])){if(e.some(t=>t.length!==e[0].length))throw Error(`Unable to create tensor, you should probably activate truncation and/or padding with 'padding=True' and/or 'truncation=True' to have batched tensors with the same length.`);return new U(`int64`,BigInt64Array.from(e.flat().map(e=>BigInt(e))),[e.length,e[0].length])}else return new U(`int64`,BigInt64Array.from(e.map(e=>BigInt(e))),[1,e.length])}function xh(e){return new U(`bool`,[e],[1])}var Sh={[X.DecoderOnly]:{can_generate:!0,forward:Ph,prepare_inputs:Bh},[X.DecoderOnlyWithoutHead]:{can_generate:!1,forward:Ph,prepare_inputs:Bh},[X.Seq2Seq]:{can_generate:!0,forward:Dh,prepare_inputs:Vh},[X.Vision2Seq]:{can_generate:!0,forward:Dh,prepare_inputs:Vh},[X.Musicgen]:{can_generate:!0,forward:Dh},[X.EncoderDecoder]:{can_generate:!1,forward:Dh},[X.ImageTextToText]:{can_generate:!0,forward:Lh,prepare_inputs:Hh},[X.AudioTextToText]:{can_generate:!0,forward:Ih,prepare_inputs:Hh},[X.ImageAudioTextToText]:{can_generate:!0,prepare_inputs:Hh},[X.Phi3V]:{can_generate:!0,prepare_inputs:Hh},[X.MultiModality]:{can_generate:!0},[X.AutoEncoder]:{can_generate:!1,forward:kh},[X.Chatterbox]:{can_generate:!0,forward:Oh},[X.VoxtralRealtime]:{can_generate:!0,prepare_inputs:Bh},default:{can_generate:!1,forward:Oh}};function Ch(e,t){let n=wh.get(e),r=!1,i=t?.architectures?.[0];if(i&&i!==e&&e?.endsWith(`ForCausalLM`)&&i.endsWith(`ForConditionalGeneration`)){let e=wh.get(i);e!==void 0&&(n=e,r=!0)}let a=Sh[n]??Sh.default,o=ph[n]??ph.default;return{typeConfig:{...a,...o},textOnly:r,modelType:n}}var wh=new Map,Th=new Map,Eh=new Map,Z=class extends ni{main_input_name=`input_ids`;forward_params=[`input_ids`,`attention_mask`];_return_dict_in_generate_keys=null;constructor(e,t,n){super(),this.config=e,this.sessions=t,this.configs=n;let{typeConfig:r}=Ch(Eh.get(this.constructor),e);this.can_generate=r.can_generate,this._forward=r.forward,this._prepare_inputs_for_generation=r.prepare_inputs,this.can_generate&&this.forward_params.push(`past_key_values`),this.custom_config=this.config[`transformers.js_config`]??{}}async dispose(){let e=[];for(let t of Object.values(this.sessions))e.push(t.release?.());return await Promise.all(e)}static async from_pretrained(e,{progress_callback:t=null,config:n=null,cache_dir:r=null,local_files_only:i=!1,revision:a=`main`,model_file_name:o=null,subfolder:s=`onnx`,device:c=null,dtype:l=null,use_external_data_format:u=null,session_options:d={}}={}){let f={progress_callback:t,config:n,cache_dir:r,local_files_only:i,revision:a,model_file_name:o,subfolder:s,device:c,dtype:l,use_external_data_format:u,session_options:d},p=Eh.get(this);n=f.config=await Dm.from_pretrained(e,f);let{typeConfig:m,textOnly:h,modelType:g}=Ch(p,n);if(g===void 0){let e=p??n?.model_type;e!==`custom`&&N.warn(`Model type for '${e}' not found, assuming encoder-only architecture. Please report this at ${Fu}.`)}if(t&&!(t instanceof ii)){let r={};try{let t=await _h(e,{config:n,dtype:l,device:c,model_file_name:o});(await Promise.all(t.map(t=>Ks(e,t,f)))).forEach((e,n)=>{if(e.exists){let i=t[n]===`config.json`;r[t[n]]={loaded:i?e.size??0:0,total:e.size??0}}})}catch(e){N.warn(`Unable to fetch model file metadata for total progress tracking: ${e}`)}Object.keys(r).length>0&&(f.progress_callback=new ii(t,r))}let _=[Nm(e,m.sessions(n,f,h),f,m.cache_sessions)];m.optional_configs&&_.push(Kh(e,m.optional_configs,f));let v=await Promise.all(_);return new this(n,...v)}async _call(e){return await this.forward(e)}async forward(e){return await this._forward(this,e)}get generation_config(){return this.configs?.generation_config??null}_get_logits_processor(e,t,n=null){let r=new Wm;if(e.repetition_penalty!==null&&e.repetition_penalty!==1&&r.push(new Zm(e.repetition_penalty)),e.no_repeat_ngram_size!==null&&e.no_repeat_ngram_size>0&&r.push(new Xm(e.no_repeat_ngram_size)),e.bad_words_ids!==null&&r.push(new eh(e.bad_words_ids,e.eos_token_id)),e.min_length!==null&&e.eos_token_id!==null&&e.min_length>0&&r.push(new Qm(e.min_length,e.eos_token_id)),e.min_new_tokens!==null&&e.eos_token_id!==null&&e.min_new_tokens>0&&r.push(new $m(t,e.min_new_tokens,e.eos_token_id)),e.forced_bos_token_id!==null&&r.push(new Gm(e.forced_bos_token_id)),e.forced_eos_token_id!==null&&r.push(new Km(e.max_length,e.forced_eos_token_id)),e.suppress_tokens!==null&&r.push(new qm(e.suppress_tokens)),e.begin_suppress_tokens!==null){let n=t>1||e.forced_bos_token_id===null?t:t+1;r.push(new Jm(e.begin_suppress_tokens,n))}return e.guidance_scale!==null&&e.guidance_scale>1&&r.push(new th(e.guidance_scale)),e.temperature===0&&e.do_sample&&(N.warn("`do_sample` changed to false because `temperature: 0` implies greedy sampling (always selecting the most likely token), which is incompatible with `do_sample: true`."),e.do_sample=!1),e.do_sample&&e.temperature!==null&&e.temperature!==1&&r.push(new nh(e.temperature)),n!==null&&r.extend(n),r}_prepare_generation_config(e,t,n=rh){let r={...this.config};for(let e of[`decoder`,`generator`,`text_config`])e in r&&Object.assign(r,r[e]);let i=new n(r);return Object.assign(i,this.generation_config??{}),e&&Object.assign(i,e),t&&Object.assign(i,di(t,Object.getOwnPropertyNames(i))),i}_get_stopping_criteria(e,t=null){let n=new ah;return e.max_length!==null&&n.push(new oh(e.max_length,this.config.max_position_embeddings??null)),e.eos_token_id!==null&&n.push(new sh(e.eos_token_id)),t&&n.extend(t),n}_validate_model_class(){if(!this.can_generate){let e=[vh.MODEL_FOR_CAUSAL_LM_MAPPING_NAMES,vh.MODEL_FOR_VISION_2_SEQ_MAPPING_NAMES,vh.MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING_NAMES,vh.MODEL_FOR_SPEECH_SEQ_2_SEQ_MAPPING_NAMES].filter(Boolean),t=Eh.get(this.constructor),n=new Set,r=this.config.model_type;for(let t of e){let e=t?.get(r);e&&n.add(e)}let i=`The current model class (${t}) is not compatible with \`.generate()\`, as it doesn't have a language model head.`;throw n.size>0&&(i+=` Please use the following class instead: ${[...n].join(`, `)}`),Error(i)}}prepare_inputs_for_generation(...e){if(!this._prepare_inputs_for_generation)throw Error(`prepare_inputs_for_generation is not implemented for this model.`);return this._prepare_inputs_for_generation(this,...e)}_update_model_kwargs_for_generation({generated_input_ids:e,outputs:t,model_inputs:n,is_encoder_decoder:r}){return n.past_key_values=Ah(t,n.past_key_values),n.input_ids=new U(`int64`,e.flat(),[e.length,1]),r?`decoder_attention_mask`in n&&(n.decoder_attention_mask=fl([n.decoder_attention_mask,xl([n.decoder_attention_mask.dims[0],1])],1)):n.attention_mask=fl([n.attention_mask,xl([n.attention_mask.dims[0],1])],1),n.position_ids=null,n}_prepare_model_inputs({inputs:e,bos_token_id:t,model_kwargs:n}){let r=di(n,this.forward_params),i=this.main_input_name;if(i in r){if(e)throw Error("`inputs`: {inputs}` were passed alongside {input_name} which is not allowed. Make sure to either pass {inputs} or {input_name}=...")}else r[i]=e;return{inputs_tensor:r[i],model_inputs:r,model_input_name:i}}async _prepare_encoder_decoder_kwargs_for_generation({inputs_tensor:e,model_inputs:t,model_input_name:n,generation_config:r}){if(this.sessions.model.inputNames.includes(`inputs_embeds`)&&!t.inputs_embeds&&`_prepare_inputs_embeds`in this){let{input_ids:e,pixel_values:n,attention_mask:r,...i}=t,a=await this._prepare_inputs_embeds(t);t={...i,...di(a,[`inputs_embeds`,`attention_mask`])}}let{last_hidden_state:i}=await Oh(this,t);if(r.guidance_scale!==null&&r.guidance_scale>1)i=fl([i,bl(i,0)],0),`attention_mask`in t&&(t.attention_mask=fl([t.attention_mask,wl(t.attention_mask)],0));else if(t.decoder_input_ids){let e=bh(t.decoder_input_ids).dims[0];if(e!==i.dims[0]){if(i.dims[0]!==1)throw Error(`The encoder outputs have a different batch size (${i.dims[0]}) than the decoder inputs (${e}).`);i=fl(Array.from({length:e},()=>i),0)}}return t.encoder_outputs=i,t}_prepare_decoder_input_ids_for_generation({batch_size:e,model_input_name:t,model_kwargs:n,decoder_start_token_id:r,bos_token_id:i,generation_config:a}){let{decoder_input_ids:o,...s}=n;if(!(o instanceof U)){if(o)Array.isArray(o[0])||(o=Array.from({length:e},()=>o));else if(r??=i,this.config.model_type===`musicgen`)o=Array.from({length:e*this.config.decoder.num_codebooks},()=>[r]);else if(Array.isArray(r)){if(r.length!==e)throw Error(`\`decoder_start_token_id\` expcted to have length ${e} but got ${r.length}`);o=r}else o=Array.from({length:e},()=>[r]);o=bh(o)}return s.decoder_attention_mask=Sl(o),{input_ids:o,model_inputs:s}}async generate({inputs:e=null,generation_config:t=null,logits_processor:n=null,stopping_criteria:r=null,streamer:i=null,...a}){this._validate_model_class(),t=this._prepare_generation_config(t,a);let{inputs_tensor:o,model_inputs:s,model_input_name:c}=this._prepare_model_inputs({inputs:e,model_kwargs:a}),l=this.config.is_encoder_decoder;l&&(`encoder_outputs`in s||(s=await this._prepare_encoder_decoder_kwargs_for_generation({inputs_tensor:o,model_inputs:s,model_input_name:c,generation_config:t})));let u;l?{input_ids:u,model_inputs:s}=this._prepare_decoder_input_ids_for_generation({batch_size:s[c].dims.at(0),model_input_name:c,model_kwargs:s,decoder_start_token_id:t.decoder_start_token_id,bos_token_id:t.bos_token_id,generation_config:t}):u=s[c];let d=u.dims.at(-1);t.max_new_tokens!==null&&(t.max_length=d+t.max_new_tokens);let f=this._get_logits_processor(t,d,n),p=this._get_stopping_criteria(t,r),m=s[c].dims.at(0),h=ch.getSampler(t),g=Array(m).fill(0),_=u.tolist();i&&i.put(_);let v,y={},b={};for(;;){if(s=this.prepare_inputs_for_generation(_,s,t),v=await this.forward(s),t.return_dict_in_generate)if(t.output_attentions){let e=jh(v);for(let t in e)t in y||(y[t]=[]),y[t].push(e[t])}else this._return_dict_in_generate_keys&&Object.assign(b,di(v,this._return_dict_in_generate_keys));let e=f(_,v.logits.slice(null,-1,null).to(`float32`)),n=[];for(let t=0;t<e.dims.at(0);++t){let r=e[t],i=await h(r);for(let[e,r]of i){let i=BigInt(e);g[t]+=r,_[t].push(i),n.push([i]);break}}if(i&&i.put(n),p(_).every(e=>e))break;s=this._update_model_kwargs_for_generation({generated_input_ids:n,outputs:v,model_inputs:s,is_encoder_decoder:l})}i&&i.end();let x=new U(`int64`,_.flat(),[_.length,_[0].length]),ee=Ah(v,s.past_key_values),S=new Set(Object.values(ee));for(let e of Object.values(v))e.location===`gpu-buffer`&&!S.has(e)&&e.dispose();return`past_key_values`in a||t.return_dict_in_generate||await ee.dispose(),t.return_dict_in_generate?{sequences:x,past_key_values:ee,...y,...b}:x}async _encode_input(e,t,n){if(!Object.hasOwn(this.sessions,e))throw Error(`Model does not have a ${e} session.`);let r=this.sessions[e];return(await J(r,di(t,r.inputNames)))[n]}async encode_image(e){return this._encode_input(`vision_encoder`,e,`image_features`)}async encode_text(e){return this._encode_input(`embed_tokens`,e,`inputs_embeds`)}async encode_audio(e){return this._encode_input(`audio_encoder`,e,`audio_features`)}};async function Dh(e,t){let{encoder_outputs:n,input_ids:r,decoder_input_ids:i,decoder_attention_mask:a,...o}=t;return n||=(await Oh(e,di(t,e.sessions.model.inputNames))).last_hidden_state,o.input_ids=i,o.encoder_hidden_states=n,e.sessions.decoder_model_merged.inputNames.includes(`encoder_attention_mask`)&&(o.encoder_attention_mask=t.attention_mask),a&&!o.attention_mask&&(o.attention_mask=a),await Ph(e,o,!0)}async function Oh(e,t){let n=e.sessions.model,r=di(t,n.inputNames);if(n.inputNames.includes(`inputs_embeds`)&&!r.inputs_embeds){if(!t.input_ids)throw Error("Both `input_ids` and `inputs_embeds` are missing in the model inputs.");r.inputs_embeds=await e.encode_text({input_ids:t.input_ids})}if(n.inputNames.includes(`token_type_ids`)&&!r.token_type_ids){if(!r.input_ids)throw Error("Both `input_ids` and `token_type_ids` are missing in the model inputs.");r.token_type_ids=wl(r.input_ids)}if(n.inputNames.includes(`pixel_mask`)&&!r.pixel_mask){if(!r.pixel_values)throw Error("Both `pixel_values` and `pixel_mask` are missing in the model inputs.");let e=r.pixel_values.dims;r.pixel_mask=xl([e[0],e[2],e[3]])}return await J(n,r)}async function kh(e,t){let n=await e.encode(t);return await e.decode(n)}function Ah(e,t){let n=Object.create(null);for(let r in e)if(r.startsWith(`present`)){let i=r.replace(`present_ssm`,`past_ssm`).replace(`present_conv`,`past_conv`).replace(`present_recurrent`,`past_recurrent`).replace(`present`,`past_key_values`);r.includes(`encoder`)&&t?n[i]=t[i]:n[i]=e[r]}return t?(t.update(n),t):new fh(n)}function jh(e){let t={};for(let n of[`cross_attentions`,`encoder_attentions`,`decoder_attentions`])for(let r in e)r.startsWith(n)&&(n in t||(t[n]=[]),t[n].push(e[r]));return t}function Mh(e,t){return e.map(e=>typeof e==`number`?e:t[e]??0)}function Nh(e,t,n){if(n&&Object.keys(n).length>0)return Object.assign(t,n),n;let r=e.sessions.decoder_model_merged??e.sessions.model,i=(t[e.main_input_name]??t.attention_mask)?.dims?.[0]??1,a=wm(e.config),o=e.config?.normalized_config?.num_heads,s={batch_size:i};typeof o==`number`&&(s[`batch_size x num_heads`]=i*o);let c=Object.create(null);for(let e of r.inputMetadata){if(!a.has(e.name))continue;let n=Mh(e.shape,s),r=n.reduce((e,t)=>e*t,1),i=$c[e.type],o=new U(e.type,new i(r),n);t[e.name]=o,c[e.name]=o}return n?(n.update(c),n):new fh(c)}async function Ph(e,t,n=!1){let r=e.sessions[n?`decoder_model_merged`:`model`],{past_key_values:i,...a}=t;return r.inputNames.includes(`use_cache_branch`)&&(a.use_cache_branch=xh(i!=null&&Object.keys(i).length>0)),r.inputNames.includes(`position_ids`)&&a.attention_mask&&!a.position_ids&&(a.position_ids=zh(a,i,+!![`paligemma`,`gemma3_text`,`gemma3`].includes(e.config.model_type))),r.inputNames.includes(`num_logits_to_keep`)&&!a.num_logits_to_keep&&(a.num_logits_to_keep=new U(`int64`,[0n],[])),Nh(e,a,i),await J(r,di(a,r.inputNames))}async function Fh(e,{encode_function:t,merge_function:n,modality_input_names:r,modality_output_name:i,input_ids:a=null,attention_mask:o=null,position_ids:s=null,inputs_embeds:c=null,past_key_values:l=null,generation_config:u=null,logits_processor:d=null,...f}){if(!c){c=await e.encode_text({input_ids:a,...f});let s=di(f,r);if(Object.keys(s).length>0){if(a.dims[1]!==1){let e=await t({...s,...f});({inputs_embeds:c,attention_mask:o}=n({[i]:e,inputs_embeds:c,input_ids:a,attention_mask:o}))}else if(l&&a.dims[1]===1){let e=a.dims[1],t=l.get_seq_length();o=fl([xl([a.dims[0],t]),o.slice(null,[o.dims[1]-e,o.dims[1]])],1)}}}if(!s&&[`qwen2_vl`,`qwen2_vl_text`,`qwen2_5_vl`,`qwen2_5_vl_text`,`qwen3_vl`,`qwen3_vl_text`,`qwen3_vl_moe`,`qwen3_vl_moe_text`,`qwen3_5`,`qwen3_5_text`,`qwen3_5_moe`,`qwen3_5_moe_text`,`glm_ocr`,`glm_ocr_text`].includes(e.config.model_type)){let{image_grid_thw:t,video_grid_thw:n}=f;[s]=e.get_rope_index(a,t,n,o)}return await Ph(e,{inputs_embeds:c,past_key_values:l,attention_mask:o,position_ids:s,generation_config:u,logits_processor:d},!0)}async function Ih(e,t){return await Fh(e,{...t,modality_input_names:[`audio_values`,`input_features`],modality_output_name:`audio_features`,encode_function:e.encode_audio.bind(e),merge_function:e._merge_input_ids_with_audio_features.bind(e)})}async function Lh(e,t){return await Fh(e,{...t,modality_input_names:[`pixel_values`],modality_output_name:`image_features`,encode_function:e.encode_image.bind(e),merge_function:e._merge_input_ids_with_image_features.bind(e)})}function Rh(e,t=0){let[n,r]=e.dims,i=e.data,a=new BigInt64Array(i.length);for(let e=0;e<n;++e){let n=e*r,o=BigInt(t);for(let e=0;e<r;++e){let t=n+e;i[t]===0n?a[t]=BigInt(1):(a[t]=o,o+=i[t])}}return{data:a,dims:e.dims}}function zh(e,t=null,n=0){let{input_ids:r,inputs_embeds:i,attention_mask:a}=e,{data:o,dims:s}=Rh(a,n),c=new U(`int64`,o,s);if(t){let e=-(r??i).dims.at(1);c=c.slice(null,[e,null])}return c}function Bh(e,t,n,r){let i=n.past_key_values?n.past_key_values.get_seq_length():0;if((e.sessions.decoder_model_merged??e.sessions.model)?.inputNames.includes(`num_logits_to_keep`)&&!n.num_logits_to_keep&&(n.num_logits_to_keep=new U(`int64`,[1n],[])),!n.attention_mask){let e;for(let t of[`input_ids`,`inputs_embeds`,`position_ids`])if(n[t]){e=n[t].dims;break}if(!e)throw Error(`attention_mask is not provided, and unable to infer its shape from model inputs.`);n.attention_mask=xl([e[0],i+e[1]])}if(n.past_key_values){let{input_ids:e,attention_mask:t}=n;t&&t.dims[1]>e.dims[1]||i<e.dims[1]&&(n.input_ids=e.slice(null,[i,null]))}return n}function Vh(e,t,n,r){return n.past_key_values&&(t=t.map(e=>[e.at(-1)])),{...n,decoder_input_ids:bh(t)}}function Hh(e,...t){return e.config.is_encoder_decoder?Vh(e,...t):Bh(e,...t)}function Uh({modality_token_id:e,inputs_embeds:t,modality_features:n,input_ids:r,attention_mask:i}){let a=r.tolist().map(t=>t.reduce((t,n,r)=>(n==e&&t.push(r),t),[])),o=a.reduce((e,t)=>e+t.length,0),s=n.dims[0];if(o!==s)throw Error(`Number of tokens and features do not match: tokens: ${o}, features ${s}`);let c=0;for(let e=0;e<a.length;++e){let r=a[e],i=t[e];for(let e=0;e<r.length;++e)i[r[e]].data.set(n[c++].data)}return{inputs_embeds:t,attention_mask:i}}function Wh({image_token_id:e,inputs_embeds:t,image_features:n,input_ids:r,attention_mask:i}){return Uh({modality_token_id:e,inputs_embeds:t,modality_features:n,input_ids:r,attention_mask:i})}function Gh({audio_token_id:e,inputs_embeds:t,audio_features:n,input_ids:r,attention_mask:i}){return Uh({modality_token_id:e,inputs_embeds:t,modality_features:n,input_ids:r,attention_mask:i})}async function Kh(e,t,n){return Object.fromEntries(await Promise.all(Object.keys(t).map(async r=>[r,await rc(e,t[r],!1,n)])))}var qh={};Tr(qh,{ASTForAudioClassification:()=>ug,ASTModel:()=>lg,ASTPreTrainedModel:()=>cg,AfmoeForCausalLM:()=>ig,AfmoeModel:()=>rg,AfmoePreTrainedModel:()=>ng,AlbertForMaskedLM:()=>Qh,AlbertForQuestionAnswering:()=>Zh,AlbertForSequenceClassification:()=>Xh,AlbertModel:()=>Yh,AlbertPreTrainedModel:()=>Jh,ApertusForCausalLM:()=>tg,ApertusModel:()=>eg,ApertusPreTrainedModel:()=>$h,ArceeForCausalLM:()=>sg,ArceeModel:()=>og,ArceePreTrainedModel:()=>ag,BartForConditionalGeneration:()=>pg,BartForSequenceClassification:()=>mg,BartModel:()=>fg,BartPretrainedModel:()=>dg,BeitForImageClassification:()=>_g,BeitModel:()=>gg,BeitPreTrainedModel:()=>hg,BertForMaskedLM:()=>bg,BertForQuestionAnswering:()=>Cg,BertForSequenceClassification:()=>xg,BertForTokenClassification:()=>Sg,BertModel:()=>yg,BertPreTrainedModel:()=>vg,BlenderbotForConditionalGeneration:()=>Eg,BlenderbotModel:()=>Tg,BlenderbotPreTrainedModel:()=>wg,BlenderbotSmallForConditionalGeneration:()=>kg,BlenderbotSmallModel:()=>Og,BlenderbotSmallPreTrainedModel:()=>Dg,BloomForCausalLM:()=>Mg,BloomModel:()=>jg,BloomPreTrainedModel:()=>Ag,CHMv2ForDepthEstimation:()=>Kg,CHMv2PreTrainedModel:()=>Gg,CLIPModel:()=>Qg,CLIPPreTrainedModel:()=>Zg,CLIPSegForImageSegmentation:()=>a_,CLIPSegModel:()=>i_,CLIPSegPreTrainedModel:()=>r_,CLIPTextModel:()=>$g,CLIPTextModelWithProjection:()=>e_,CLIPVisionModel:()=>t_,CLIPVisionModelWithProjection:()=>n_,CamembertForMaskedLM:()=>Fg,CamembertForQuestionAnswering:()=>Rg,CamembertForSequenceClassification:()=>Ig,CamembertForTokenClassification:()=>Lg,CamembertModel:()=>Pg,CamembertPreTrainedModel:()=>Ng,ChatterboxModel:()=>Hg,ChatterboxPreTrainedModel:()=>Vg,ChineseCLIPModel:()=>Wg,ChineseCLIPPreTrainedModel:()=>Ug,ClapAudioModelWithProjection:()=>Xg,ClapModel:()=>Jg,ClapPreTrainedModel:()=>qg,ClapTextModelWithProjection:()=>Yg,CodeGenForCausalLM:()=>c_,CodeGenModel:()=>s_,CodeGenPreTrainedModel:()=>o_,Cohere2ForCausalLM:()=>m_,Cohere2Model:()=>p_,Cohere2PreTrainedModel:()=>f_,CohereAsrForConditionalGeneration:()=>__,CohereAsrModel:()=>g_,CohereAsrPreTrainedModel:()=>h_,CohereForCausalLM:()=>d_,CohereModel:()=>u_,CoherePreTrainedModel:()=>l_,ConvBertForMaskedLM:()=>b_,ConvBertForQuestionAnswering:()=>C_,ConvBertForSequenceClassification:()=>x_,ConvBertForTokenClassification:()=>S_,ConvBertModel:()=>y_,ConvBertPreTrainedModel:()=>v_,ConvNextForImageClassification:()=>E_,ConvNextModel:()=>T_,ConvNextPreTrainedModel:()=>w_,ConvNextV2ForImageClassification:()=>k_,ConvNextV2Model:()=>O_,ConvNextV2PreTrainedModel:()=>D_,DFineForObjectDetection:()=>I_,DFineModel:()=>F_,DFinePreTrainedModel:()=>P_,DINOv3ConvNextModel:()=>Ev,DINOv3ConvNextPreTrainedModel:()=>Tv,DINOv3ViTModel:()=>Ov,DINOv3ViTPreTrainedModel:()=>Dv,DPTForDepthEstimation:()=>zv,DPTModel:()=>Rv,DPTPreTrainedModel:()=>Lv,DacDecoderModel:()=>H_,DacDecoderOutput:()=>R_,DacEncoderModel:()=>V_,DacEncoderOutput:()=>L_,DacModel:()=>B_,DacPreTrainedModel:()=>z_,DebertaForMaskedLM:()=>G_,DebertaForQuestionAnswering:()=>J_,DebertaForSequenceClassification:()=>K_,DebertaForTokenClassification:()=>q_,DebertaModel:()=>W_,DebertaPreTrainedModel:()=>U_,DebertaV2ForMaskedLM:()=>ev,DebertaV2ForQuestionAnswering:()=>rv,DebertaV2ForSequenceClassification:()=>tv,DebertaV2ForTokenClassification:()=>nv,DebertaV2Model:()=>$_,DebertaV2PreTrainedModel:()=>Q_,DecisionTransformerModel:()=>av,DecisionTransformerPreTrainedModel:()=>iv,DeepseekV3ForCausalLM:()=>Z_,DeepseekV3Model:()=>X_,DeepseekV3PreTrainedModel:()=>Y_,DeiTForImageClassification:()=>cv,DeiTModel:()=>sv,DeiTPreTrainedModel:()=>ov,DepthAnythingForDepthEstimation:()=>uv,DepthAnythingPreTrainedModel:()=>lv,DepthProForDepthEstimation:()=>fv,DepthProPreTrainedModel:()=>dv,DetrForObjectDetection:()=>hv,DetrForSegmentation:()=>gv,DetrModel:()=>mv,DetrObjectDetectionOutput:()=>_v,DetrPreTrainedModel:()=>pv,DetrSegmentationOutput:()=>vv,Dinov2ForImageClassification:()=>xv,Dinov2Model:()=>bv,Dinov2PreTrainedModel:()=>yv,Dinov2WithRegistersForImageClassification:()=>wv,Dinov2WithRegistersModel:()=>Cv,Dinov2WithRegistersPreTrainedModel:()=>Sv,DistilBertForMaskedLM:()=>Pv,DistilBertForQuestionAnswering:()=>Nv,DistilBertForSequenceClassification:()=>jv,DistilBertForTokenClassification:()=>Mv,DistilBertModel:()=>Av,DistilBertPreTrainedModel:()=>kv,DonutSwinModel:()=>Iv,DonutSwinPreTrainedModel:()=>Fv,EdgeTamModel:()=>gT,EfficientNetForImageClassification:()=>Hv,EfficientNetModel:()=>Vv,EfficientNetPreTrainedModel:()=>Bv,ElectraForMaskedLM:()=>Gv,ElectraForQuestionAnswering:()=>Jv,ElectraForSequenceClassification:()=>Kv,ElectraForTokenClassification:()=>qv,ElectraModel:()=>Wv,ElectraPreTrainedModel:()=>Uv,Ernie4_5ForCausalLM:()=>Zv,Ernie4_5Model:()=>Xv,Ernie4_5PretrainedModel:()=>Yv,EsmForMaskedLM:()=>ey,EsmForSequenceClassification:()=>ty,EsmForTokenClassification:()=>ny,EsmModel:()=>$v,EsmPreTrainedModel:()=>Qv,EuroBertForMaskedLM:()=>ay,EuroBertForSequenceClassification:()=>oy,EuroBertForTokenClassification:()=>sy,EuroBertModel:()=>iy,EuroBertPreTrainedModel:()=>ry,ExaoneForCausalLM:()=>uy,ExaoneModel:()=>ly,ExaonePreTrainedModel:()=>cy,FalconForCausalLM:()=>py,FalconH1ForCausalLM:()=>gy,FalconH1Model:()=>hy,FalconH1PreTrainedModel:()=>my,FalconModel:()=>fy,FalconPreTrainedModel:()=>dy,FastViTForImageClassification:()=>yy,FastViTModel:()=>vy,FastViTPreTrainedModel:()=>_y,Florence2ForConditionalGeneration:()=>xy,Florence2PreTrainedModel:()=>by,GLPNForDepthEstimation:()=>tb,GLPNModel:()=>eb,GLPNPreTrainedModel:()=>$y,GPT2LMHeadModel:()=>gb,GPT2Model:()=>hb,GPT2PreTrainedModel:()=>mb,GPTBigCodeForCausalLM:()=>ib,GPTBigCodeModel:()=>rb,GPTBigCodePreTrainedModel:()=>nb,GPTJForCausalLM:()=>yb,GPTJModel:()=>vb,GPTJPreTrainedModel:()=>_b,GPTNeoForCausalLM:()=>sb,GPTNeoModel:()=>ob,GPTNeoPreTrainedModel:()=>ab,GPTNeoXForCausalLM:()=>ub,GPTNeoXModel:()=>lb,GPTNeoXPreTrainedModel:()=>cb,Gemma2ForCausalLM:()=>Dy,Gemma2Model:()=>Ey,Gemma2PreTrainedModel:()=>Ty,Gemma3ForCausalLM:()=>Fy,Gemma3ForConditionalGeneration:()=>Py,Gemma3Model:()=>Ny,Gemma3PreTrainedModel:()=>My,Gemma3nForCausalLM:()=>Ry,Gemma3nForConditionalGeneration:()=>Ly,Gemma3nPreTrainedModel:()=>Iy,Gemma4ForCausalLM:()=>By,Gemma4ForConditionalGeneration:()=>zy,GemmaForCausalLM:()=>wy,GemmaModel:()=>Cy,GemmaPreTrainedModel:()=>Sy,GlmForCausalLM:()=>Uy,GlmModel:()=>Hy,GlmMoeDsaForCausalLM:()=>Ky,GlmMoeDsaModel:()=>Gy,GlmMoeDsaPreTrainedModel:()=>Wy,GlmOcrForConditionalGeneration:()=>Qy,GlmPreTrainedModel:()=>Vy,GptOssForCausalLM:()=>pb,GptOssModel:()=>fb,GptOssPreTrainedModel:()=>db,GraniteForCausalLM:()=>Sb,GraniteModel:()=>xb,GraniteMoeHybridForCausalLM:()=>Tb,GraniteMoeHybridModel:()=>wb,GraniteMoeHybridPreTrainedModel:()=>Cb,GranitePreTrainedModel:()=>bb,GraniteSpeechForConditionalGeneration:()=>Ob,GroundingDinoForObjectDetection:()=>Ab,GroundingDinoPreTrainedModel:()=>kb,GroupViTModel:()=>Mb,GroupViTPreTrainedModel:()=>jb,HeliumForCausalLM:()=>Fb,HeliumModel:()=>Pb,HeliumPreTrainedModel:()=>Nb,HieraForImageClassification:()=>Rb,HieraModel:()=>Lb,HieraPreTrainedModel:()=>Ib,HubertForCTC:()=>Kb,HubertForSequenceClassification:()=>qb,HubertModel:()=>Gb,HubertPreTrainedModel:()=>Wb,HunYuanDenseV1ForCausalLM:()=>Xb,HunYuanDenseV1Model:()=>Yb,HunYuanDenseV1PreTrainedModel:()=>Jb,IJepaForImageClassification:()=>ex,IJepaModel:()=>$b,IJepaPreTrainedModel:()=>Qb,Idefics3ForConditionalGeneration:()=>Zb,JAISLMHeadModel:()=>rx,JAISModel:()=>nx,JAISPreTrainedModel:()=>tx,JinaCLIPModel:()=>ax,JinaCLIPPreTrainedModel:()=>ix,JinaCLIPTextModel:()=>ox,JinaCLIPVisionModel:()=>sx,Lfm2ForCausalLM:()=>ux,Lfm2Model:()=>lx,Lfm2MoeForCausalLM:()=>mx,Lfm2MoeModel:()=>px,Lfm2MoePreTrainedModel:()=>fx,Lfm2PreTrainedModel:()=>cx,Lfm2VlForConditionalGeneration:()=>hx,LightOnOcrForConditionalGeneration:()=>dx,LiteWhisperForConditionalGeneration:()=>TD,Llama4ForCausalLM:()=>bx,Llama4PreTrainedModel:()=>yx,LlamaForCausalLM:()=>vx,LlamaModel:()=>_x,LlamaPreTrainedModel:()=>gx,LlavaForConditionalGeneration:()=>ky,LlavaOnevisionForConditionalGeneration:()=>ky,LlavaPreTrainedModel:()=>Oy,LlavaQwen2ForCausalLM:()=>jy,LongT5ForConditionalGeneration:()=>Cx,LongT5Model:()=>Sx,LongT5PreTrainedModel:()=>xx,M2M100ForConditionalGeneration:()=>Ex,M2M100Model:()=>Tx,M2M100PreTrainedModel:()=>wx,MBartForCausalLM:()=>Lx,MBartForConditionalGeneration:()=>Fx,MBartForSequenceClassification:()=>Ix,MBartModel:()=>Px,MBartPreTrainedModel:()=>Nx,MPNetForMaskedLM:()=>KS,MPNetForQuestionAnswering:()=>YS,MPNetForSequenceClassification:()=>qS,MPNetForTokenClassification:()=>JS,MPNetModel:()=>GS,MPNetPreTrainedModel:()=>WS,MT5ForConditionalGeneration:()=>tC,MT5Model:()=>eC,MT5PreTrainedModel:()=>$S,MarianMTModel:()=>kx,MarianModel:()=>Ox,MarianPreTrainedModel:()=>Dx,MaskFormerForInstanceSegmentation:()=>Mx,MaskFormerModel:()=>jx,MaskFormerPreTrainedModel:()=>Ax,Metric3DForDepthEstimation:()=>zx,Metric3DPreTrainedModel:()=>Rx,Metric3Dv2ForDepthEstimation:()=>Vx,Metric3Dv2PreTrainedModel:()=>Bx,MgpstrForSceneTextRecognition:()=>Wx,MgpstrModelOutput:()=>Hx,MgpstrPreTrainedModel:()=>Ux,MimiDecoderModel:()=>Xx,MimiDecoderOutput:()=>Kx,MimiEncoderModel:()=>Yx,MimiEncoderOutput:()=>Gx,MimiModel:()=>Jx,MimiPreTrainedModel:()=>qx,Mistral4ForCausalLM:()=>nS,Mistral4Model:()=>tS,Mistral4PreTrainedModel:()=>eS,MistralForCausalLM:()=>$x,MistralModel:()=>Qx,MistralPreTrainedModel:()=>Zx,MobileBertForMaskedLM:()=>aS,MobileBertForQuestionAnswering:()=>sS,MobileBertForSequenceClassification:()=>oS,MobileBertModel:()=>iS,MobileBertPreTrainedModel:()=>rS,MobileLLMForCausalLM:()=>uS,MobileLLMModel:()=>lS,MobileLLMPreTrainedModel:()=>cS,MobileNetV1ForImageClassification:()=>pS,MobileNetV1ForSemanticSegmentation:()=>mS,MobileNetV1Model:()=>fS,MobileNetV1PreTrainedModel:()=>dS,MobileNetV2ForImageClassification:()=>_S,MobileNetV2ForSemanticSegmentation:()=>vS,MobileNetV2Model:()=>gS,MobileNetV2PreTrainedModel:()=>hS,MobileNetV3ForImageClassification:()=>xS,MobileNetV3ForSemanticSegmentation:()=>SS,MobileNetV3Model:()=>bS,MobileNetV3PreTrainedModel:()=>yS,MobileNetV4ForImageClassification:()=>TS,MobileNetV4ForSemanticSegmentation:()=>ES,MobileNetV4Model:()=>wS,MobileNetV4PreTrainedModel:()=>CS,MobileViTForImageClassification:()=>kS,MobileViTModel:()=>OS,MobileViTPreTrainedModel:()=>DS,MobileViTV2ForImageClassification:()=>MS,MobileViTV2Model:()=>jS,MobileViTV2PreTrainedModel:()=>AS,ModernBertDecoderForCausalLM:()=>BS,ModernBertDecoderModel:()=>zS,ModernBertDecoderPreTrainedModel:()=>RS,ModernBertForMaskedLM:()=>FS,ModernBertForSequenceClassification:()=>IS,ModernBertForTokenClassification:()=>LS,ModernBertModel:()=>PS,ModernBertPreTrainedModel:()=>NS,Moondream1ForConditionalGeneration:()=>Ay,MoonshineForConditionalGeneration:()=>US,MoonshineModel:()=>HS,MoonshinePreTrainedModel:()=>VS,MptForCausalLM:()=>QS,MptModel:()=>ZS,MptPreTrainedModel:()=>XS,MultiModalityCausalLM:()=>rC,MultiModalityPreTrainedModel:()=>nC,MusicgenForCausalLM:()=>oC,MusicgenForConditionalGeneration:()=>sC,MusicgenModel:()=>aC,MusicgenPreTrainedModel:()=>iC,NanoChatForCausalLM:()=>uC,NanoChatModel:()=>lC,NanoChatPreTrainedModel:()=>cC,NemotronHForCausalLM:()=>pC,NemotronHModel:()=>fC,NemotronHPreTrainedModel:()=>dC,NeoBertForMaskedLM:()=>gC,NeoBertForQuestionAnswering:()=>yC,NeoBertForSequenceClassification:()=>_C,NeoBertForTokenClassification:()=>vC,NeoBertModel:()=>hC,NeoBertPreTrainedModel:()=>mC,NomicBertModel:()=>xC,NomicBertPreTrainedModel:()=>bC,OPTForCausalLM:()=>HC,OPTModel:()=>VC,OPTPreTrainedModel:()=>BC,Olmo2ForCausalLM:()=>DC,Olmo2Model:()=>EC,Olmo2PreTrainedModel:()=>TC,Olmo3ForCausalLM:()=>AC,Olmo3Model:()=>kC,Olmo3PreTrainedModel:()=>OC,OlmoForCausalLM:()=>wC,OlmoHybridForCausalLM:()=>NC,OlmoHybridModel:()=>MC,OlmoHybridPreTrainedModel:()=>jC,OlmoModel:()=>CC,OlmoPreTrainedModel:()=>SC,OpenAIPrivacyFilterForTokenClassification:()=>IC,OpenAIPrivacyFilterModel:()=>FC,OpenAIPrivacyFilterPreTrainedModel:()=>PC,OpenELMForCausalLM:()=>zC,OpenELMModel:()=>RC,OpenELMPreTrainedModel:()=>LC,OwlViTForObjectDetection:()=>JC,OwlViTModel:()=>qC,OwlViTPreTrainedModel:()=>KC,Owlv2ForObjectDetection:()=>GC,Owlv2Model:()=>WC,Owlv2PreTrainedModel:()=>UC,PaliGemmaForConditionalGeneration:()=>YC,ParakeetForCTC:()=>ZC,ParakeetPreTrainedModel:()=>XC,PatchTSMixerForPrediction:()=>ew,PatchTSMixerModel:()=>$C,PatchTSMixerPreTrainedModel:()=>QC,PatchTSTForPrediction:()=>rw,PatchTSTModel:()=>nw,PatchTSTPreTrainedModel:()=>tw,Phi3ForCausalLM:()=>lw,Phi3Model:()=>cw,Phi3PreTrainedModel:()=>sw,Phi3VForCausalLM:()=>dw,Phi3VPreTrainedModel:()=>uw,PhiForCausalLM:()=>ow,PhiModel:()=>aw,PhiPreTrainedModel:()=>iw,PreTrainedModel:()=>Z,PvtForImageClassification:()=>mw,PvtModel:()=>pw,PvtPreTrainedModel:()=>fw,PyAnnoteForAudioFrameClassification:()=>_w,PyAnnoteModel:()=>gw,PyAnnotePreTrainedModel:()=>hw,Qwen2ForCausalLM:()=>bw,Qwen2Model:()=>yw,Qwen2MoeForCausalLM:()=>Cw,Qwen2MoeModel:()=>Sw,Qwen2MoePreTrainedModel:()=>xw,Qwen2PreTrainedModel:()=>vw,Qwen2VLForCausalLM:()=>Yy,Qwen2VLForConditionalGeneration:()=>Jy,Qwen2VLPreTrainedModel:()=>qy,Qwen2_5_VLForCausalLM:()=>Zy,Qwen2_5_VLForConditionalGeneration:()=>Xy,Qwen3ForCausalLM:()=>Ew,Qwen3Model:()=>Tw,Qwen3MoeForCausalLM:()=>kw,Qwen3MoeModel:()=>Ow,Qwen3MoePreTrainedModel:()=>Dw,Qwen3NextForCausalLM:()=>Mw,Qwen3NextModel:()=>jw,Qwen3NextPreTrainedModel:()=>Aw,Qwen3PreTrainedModel:()=>ww,Qwen3VLForCausalLM:()=>Pw,Qwen3VLForConditionalGeneration:()=>Nw,Qwen3VLMoeForCausalLM:()=>Iw,Qwen3VLMoeForConditionalGeneration:()=>Fw,Qwen3_5ForCausalLM:()=>Rw,Qwen3_5ForConditionalGeneration:()=>Lw,Qwen3_5MoeForCausalLM:()=>Bw,Qwen3_5MoeForConditionalGeneration:()=>zw,RFDetrForObjectDetection:()=>Kw,RFDetrModel:()=>Gw,RFDetrObjectDetectionOutput:()=>qw,RFDetrPreTrainedModel:()=>Ww,RTDetrForObjectDetection:()=>M_,RTDetrModel:()=>j_,RTDetrObjectDetectionOutput:()=>N_,RTDetrPreTrainedModel:()=>A_,RTDetrV2ForObjectDetection:()=>cT,RTDetrV2Model:()=>sT,RTDetrV2ObjectDetectionOutput:()=>lT,RTDetrV2PreTrainedModel:()=>oT,ResNetForImageClassification:()=>Uw,ResNetModel:()=>Hw,ResNetPreTrainedModel:()=>Vw,RoFormerForMaskedLM:()=>nT,RoFormerForQuestionAnswering:()=>aT,RoFormerForSequenceClassification:()=>rT,RoFormerForTokenClassification:()=>iT,RoFormerModel:()=>tT,RoFormerPreTrainedModel:()=>eT,RobertaForMaskedLM:()=>Xw,RobertaForQuestionAnswering:()=>$w,RobertaForSequenceClassification:()=>Zw,RobertaForTokenClassification:()=>Qw,RobertaModel:()=>Yw,RobertaPreTrainedModel:()=>Jw,Sam2ImageSegmentationOutput:()=>pT,Sam2Model:()=>hT,Sam2PreTrainedModel:()=>mT,Sam3TrackerModel:()=>_T,SamImageSegmentationOutput:()=>uT,SamModel:()=>fT,SamPreTrainedModel:()=>dT,SapiensForDepthEstimation:()=>bT,SapiensForNormalEstimation:()=>xT,SapiensForSemanticSegmentation:()=>yT,SapiensPreTrainedModel:()=>vT,SegformerForImageClassification:()=>wT,SegformerForSemanticSegmentation:()=>TT,SegformerModel:()=>CT,SegformerPreTrainedModel:()=>ST,SiglipModel:()=>DT,SiglipPreTrainedModel:()=>ET,SiglipTextModel:()=>OT,SiglipVisionModel:()=>kT,SmolLM3ForCausalLM:()=>MT,SmolLM3Model:()=>jT,SmolLM3PreTrainedModel:()=>AT,SmolVLMForConditionalGeneration:()=>NT,SnacDecoderModel:()=>LT,SnacEncoderModel:()=>IT,SnacModel:()=>FT,SnacPreTrainedModel:()=>PT,SolarOpenForCausalLM:()=>BT,SolarOpenModel:()=>zT,SolarOpenPreTrainedModel:()=>RT,SpeechT5ForSpeechToText:()=>UT,SpeechT5ForTextToSpeech:()=>WT,SpeechT5HifiGan:()=>GT,SpeechT5Model:()=>HT,SpeechT5PreTrainedModel:()=>VT,SqueezeBertForMaskedLM:()=>JT,SqueezeBertForQuestionAnswering:()=>XT,SqueezeBertForSequenceClassification:()=>YT,SqueezeBertModel:()=>qT,SqueezeBertPreTrainedModel:()=>KT,StableLmForCausalLM:()=>$T,StableLmModel:()=>QT,StableLmPreTrainedModel:()=>ZT,Starcoder2ForCausalLM:()=>nE,Starcoder2Model:()=>tE,Starcoder2PreTrainedModel:()=>eE,StyleTextToSpeech2Model:()=>iE,StyleTextToSpeech2PreTrainedModel:()=>rE,SupertonicForConditionalGeneration:()=>oE,SupertonicPreTrainedModel:()=>aE,Swin2SRForImageSuperResolution:()=>pE,Swin2SRModel:()=>fE,Swin2SRPreTrainedModel:()=>dE,SwinForImageClassification:()=>lE,SwinForSemanticSegmentation:()=>uE,SwinModel:()=>cE,SwinPreTrainedModel:()=>sE,T5ForConditionalGeneration:()=>gE,T5Model:()=>hE,T5PreTrainedModel:()=>mE,TableTransformerForObjectDetection:()=>yE,TableTransformerModel:()=>vE,TableTransformerObjectDetectionOutput:()=>bE,TableTransformerPreTrainedModel:()=>_E,TrOCRForCausalLM:()=>SE,TrOCRPreTrainedModel:()=>xE,UltravoxModel:()=>Db,UltravoxPreTrainedModel:()=>Eb,UniSpeechForCTC:()=>TE,UniSpeechForSequenceClassification:()=>EE,UniSpeechModel:()=>wE,UniSpeechPreTrainedModel:()=>CE,UniSpeechSatForAudioFrameClassification:()=>jE,UniSpeechSatForCTC:()=>kE,UniSpeechSatForSequenceClassification:()=>AE,UniSpeechSatModel:()=>OE,UniSpeechSatPreTrainedModel:()=>DE,VaultGemmaForCausalLM:()=>PE,VaultGemmaModel:()=>NE,VaultGemmaPreTrainedModel:()=>ME,ViTForImageClassification:()=>RE,ViTMAEModel:()=>BE,ViTMAEPreTrainedModel:()=>zE,ViTMSNForImageClassification:()=>UE,ViTMSNModel:()=>HE,ViTMSNPreTrainedModel:()=>VE,ViTModel:()=>LE,ViTPreTrainedModel:()=>IE,VisionEncoderDecoderModel:()=>FE,VitMatteForImageMatting:()=>GE,VitMattePreTrainedModel:()=>WE,VitPoseForPoseEstimation:()=>qE,VitPosePreTrainedModel:()=>KE,VitsModel:()=>XE,VitsModelOutput:()=>JE,VitsPreTrainedModel:()=>YE,VoxtralForConditionalGeneration:()=>ZE,VoxtralRealtimeForConditionalGeneration:()=>sD,VoxtralRealtimePreTrainedModel:()=>oD,Wav2Vec2BertForCTC:()=>uD,Wav2Vec2BertForSequenceClassification:()=>dD,Wav2Vec2BertModel:()=>lD,Wav2Vec2BertPreTrainedModel:()=>cD,Wav2Vec2ForAudioFrameClassification:()=>Ub,Wav2Vec2ForCTC:()=>Vb,Wav2Vec2ForSequenceClassification:()=>Hb,Wav2Vec2Model:()=>Bb,Wav2Vec2PreTrainedModel:()=>zb,WavLMForAudioFrameClassification:()=>vD,WavLMForCTC:()=>hD,WavLMForSequenceClassification:()=>gD,WavLMForXVector:()=>_D,WavLMModel:()=>mD,WavLMPreTrainedModel:()=>pD,WeSpeakerResNetModel:()=>bD,WeSpeakerResNetPreTrainedModel:()=>yD,WhisperForConditionalGeneration:()=>wD,WhisperModel:()=>CD,WhisperPreTrainedModel:()=>SD,XLMForQuestionAnswering:()=>jD,XLMForSequenceClassification:()=>kD,XLMForTokenClassification:()=>AD,XLMModel:()=>DD,XLMPreTrainedModel:()=>ED,XLMRobertaForMaskedLM:()=>PD,XLMRobertaForQuestionAnswering:()=>LD,XLMRobertaForSequenceClassification:()=>FD,XLMRobertaForTokenClassification:()=>ID,XLMRobertaModel:()=>ND,XLMRobertaPreTrainedModel:()=>MD,XLMWithLMHeadModel:()=>OD,XVectorOutput:()=>fD,YolosForObjectDetection:()=>BD,YolosModel:()=>zD,YolosObjectDetectionOutput:()=>VD,YolosPreTrainedModel:()=>RD,YoutuForCausalLM:()=>WD,YoutuModel:()=>UD,YoutuPreTrainedModel:()=>HD});var Jh=class extends Z{},Yh=class extends Jh{},Xh=class extends Jh{async _call(e){return new Y(await super._call(e))}},Zh=class extends Jh{async _call(e){return new zm(await super._call(e))}},Qh=class extends Jh{async _call(e){return new Rm(await super._call(e))}},$h=class extends Z{},eg=class extends $h{},tg=class extends $h{},ng=class extends Z{},rg=class extends ng{},ig=class extends ng{},ag=class extends Z{},og=class extends ag{},sg=class extends ag{},cg=class extends Z{},lg=class extends cg{},ug=class extends cg{},dg=class extends Z{},fg=class extends dg{},pg=class extends dg{},mg=class extends dg{async _call(e){return new Y(await super._call(e))}},hg=class extends Z{},gg=class extends hg{},_g=class extends hg{async _call(e){return new Y(await super._call(e))}},vg=class extends Z{},yg=class extends vg{},bg=class extends vg{async _call(e){return new Rm(await super._call(e))}},xg=class extends vg{async _call(e){return new Y(await super._call(e))}},Sg=class extends vg{async _call(e){return new Lm(await super._call(e))}},Cg=class extends vg{async _call(e){return new zm(await super._call(e))}},wg=class extends Z{},Tg=class extends wg{},Eg=class extends wg{},Dg=class extends Z{},Og=class extends Dg{},kg=class extends Dg{},Ag=class extends Z{},jg=class extends Ag{},Mg=class extends Ag{},Ng=class extends Z{},Pg=class extends Ng{},Fg=class extends Ng{async _call(e){return new Rm(await super._call(e))}},Ig=class extends Ng{async _call(e){return new Y(await super._call(e))}},Lg=class extends Ng{async _call(e){return new Lm(await super._call(e))}},Rg=class extends Ng{async _call(e){return new zm(await super._call(e))}},zg=4299n,Bg=6561n,Vg=class extends Z{forward_params=[`input_ids`,`inputs_embeds`,`attention_mask`,`position_ids`,`audio_values`,`exaggeration`,`audio_features`,`audio_tokens`,`speaker_embeddings`,`speaker_features`,`past_key_values`];main_input_name=`input_ids`;_return_dict_in_generate_keys=[`audio_tokens`,`speaker_embeddings`,`speaker_features`]},Hg=class extends Vg{async encode_speech(e){return J(this.sessions.speech_encoder,{audio_values:e})}async forward({input_ids:e=null,attention_mask:t=null,audio_values:n=null,exaggeration:r=null,position_ids:i=null,inputs_embeds:a=null,past_key_values:o=null,generation_config:s=null,logits_processor:c=null,audio_features:l=null,audio_tokens:u=null,speaker_embeddings:d=null,speaker_features:f=null,...p}){let m;if(!a){let s=this.sessions.embed_tokens.inputNames,c={input_ids:e};if(s.includes(`exaggeration`)){if(!(r instanceof U)){let t=e.dims[0];if(r==null)r=yl([t],.5);else if(typeof r==`number`)r=yl([t],r);else if(Array.isArray(r))r=new U(`float32`,r,[t]);else throw Error("Unsupported type for `exaggeration` input")}c.exaggeration=r}if(s.includes(`position_ids`)&&(c.position_ids=i),{inputs_embeds:a}=await J(this.sessions.embed_tokens,c),l&&u&&d&&f&&(m={audio_features:l,audio_tokens:u,speaker_embeddings:d,speaker_features:f}),m||n)m??=await this.encode_speech(n),a=fl([m.audio_features,a],1),t=xl([a.dims[0],a.dims[1]]);else{let e=a.dims[1];if(!o||e!==1)throw Error(`Incorrect state encountered during generation.`);let n=o.get_seq_length();t=xl([a.dims[0],n+e])}}return{...await Ph(this,{inputs_embeds:a,past_key_values:o,attention_mask:t,generation_config:s,logits_processor:c},!1),...m}}prepare_inputs_for_generation(e,t,n){return!t.position_ids&&this.sessions.embed_tokens.inputNames.includes(`position_ids`)&&(t.input_ids.dims[1]===1?t.position_ids=new U(`int64`,Array.from({length:e.length},(t,n)=>e[n].length-e[n].findLastIndex(e=>e==Bg)-1),[e.length,1]):t.position_ids=new U(`int64`,t.input_ids.tolist().map(e=>{let t=0;return e.map(e=>e>=Bg?0:t++)}).flat(),t.input_ids.dims)),t.input_ids.dims[1]===1&&(delete t.audio_values,delete t.audio_features,delete t.audio_tokens,delete t.speaker_embeddings,delete t.speaker_features),Bh(this,e,t,n)}async generate(e){let{sequences:t,audio_tokens:n,speaker_embeddings:r,speaker_features:i}=await super.generate({...e,return_dict_in_generate:!0}),a=t.slice(null,[e.input_ids.dims[1],-1]),o=fl([n,a,yl([a.dims[0],3],zg)],1),{waveform:s}=await J(this.sessions.conditional_decoder,{speech_tokens:o,speaker_features:i,speaker_embeddings:r});return s}},Ug=class extends Z{},Wg=class extends Ug{},Gg=class extends Z{},Kg=class extends Gg{},qg=class extends Z{},Jg=class extends qg{},Yg=class extends qg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`text_model`})}},Xg=class extends qg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`audio_model`})}},Zg=class extends Z{},Qg=class extends Zg{},$g=class extends Zg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`text_model`})}},e_=class extends Zg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`text_model`})}},t_=class extends Zg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`vision_model`})}},n_=class extends Zg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`vision_model`})}},r_=class extends Z{},i_=class extends r_{},a_=class extends r_{},o_=class extends Z{},s_=class extends o_{},c_=class extends o_{},l_=class extends Z{},u_=class extends l_{},d_=class extends l_{},f_=class extends Z{},p_=class extends f_{},m_=class extends f_{},h_=class extends Z{requires_attention_mask=!1;main_input_name=`input_features`;forward_params=[`input_features`,`decoder_input_ids`,`decoder_attention_mask`,`past_key_values`]},g_=class extends h_{},__=class extends h_{},v_=class extends Z{},y_=class extends v_{},b_=class extends v_{async _call(e){return new Rm(await super._call(e))}},x_=class extends v_{async _call(e){return new Y(await super._call(e))}},S_=class extends v_{async _call(e){return new Lm(await super._call(e))}},C_=class extends v_{async _call(e){return new zm(await super._call(e))}},w_=class extends Z{},T_=class extends w_{},E_=class extends w_{async _call(e){return new Y(await super._call(e))}},D_=class extends Z{},O_=class extends D_{},k_=class extends D_{async _call(e){return new Y(await super._call(e))}},A_=class extends Z{},j_=class extends A_{},M_=class extends A_{async _call(e){return new N_(await super._call(e))}},N_=class extends Im{constructor({logits:e,pred_boxes:t}){super(),this.logits=e,this.pred_boxes=t}},P_=class extends Z{},F_=class extends P_{},I_=class extends P_{async _call(e){return new N_(await super._call(e))}},L_=class extends Im{constructor({audio_codes:e}){super(),this.audio_codes=e}},R_=class extends Im{constructor({audio_values:e}){super(),this.audio_values=e}},z_=class extends Z{main_input_name=`input_values`;forward_params=[`input_values`]},B_=class extends z_{async encode(e){return new L_(await J(this.sessions.encoder_model,e))}async decode(e){return new R_(await J(this.sessions.decoder_model,e))}},V_=class extends z_{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`encoder_model`})}},H_=class extends z_{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`decoder_model`})}},U_=class extends Z{},W_=class extends U_{},G_=class extends U_{async _call(e){return new Rm(await super._call(e))}},K_=class extends U_{async _call(e){return new Y(await super._call(e))}},q_=class extends U_{async _call(e){return new Lm(await super._call(e))}},J_=class extends U_{async _call(e){return new zm(await super._call(e))}},Y_=class extends Z{},X_=class extends Y_{},Z_=class extends Y_{},Q_=class extends Z{},$_=class extends Q_{},ev=class extends Q_{async _call(e){return new Rm(await super._call(e))}},tv=class extends Q_{async _call(e){return new Y(await super._call(e))}},nv=class extends Q_{async _call(e){return new Lm(await super._call(e))}},rv=class extends Q_{async _call(e){return new zm(await super._call(e))}},iv=class extends Z{},av=class extends iv{},ov=class extends Z{},sv=class extends ov{},cv=class extends ov{async _call(e){return new Y(await super._call(e))}},lv=class extends Z{},uv=class extends lv{},dv=class extends Z{},fv=class extends dv{},pv=class extends Z{},mv=class extends pv{},hv=class extends pv{async _call(e){return new _v(await super._call(e))}},gv=class extends pv{async _call(e){return new vv(await super._call(e))}},_v=class extends Im{constructor({logits:e,pred_boxes:t}){super(),this.logits=e,this.pred_boxes=t}},vv=class extends Im{constructor({logits:e,pred_boxes:t,pred_masks:n}){super(),this.logits=e,this.pred_boxes=t,this.pred_masks=n}},yv=class extends Z{},bv=class extends yv{},xv=class extends yv{async _call(e){return new Y(await super._call(e))}},Sv=class extends Z{},Cv=class extends Sv{},wv=class extends Sv{async _call(e){return new Y(await super._call(e))}},Tv=class extends Z{},Ev=class extends Tv{},Dv=class extends Z{},Ov=class extends Dv{},kv=class extends Z{},Av=class extends kv{},jv=class extends kv{async _call(e){return new Y(await super._call(e))}},Mv=class extends kv{async _call(e){return new Lm(await super._call(e))}},Nv=class extends kv{async _call(e){return new zm(await super._call(e))}},Pv=class extends kv{async _call(e){return new Rm(await super._call(e))}},Fv=class extends Z{},Iv=class extends Fv{},Lv=class extends Z{},Rv=class extends Lv{},zv=class extends Lv{},Bv=class extends Z{},Vv=class extends Bv{},Hv=class extends Bv{async _call(e){return new Y(await super._call(e))}},Uv=class extends Z{},Wv=class extends Uv{},Gv=class extends Uv{async _call(e){return new Rm(await super._call(e))}},Kv=class extends Uv{async _call(e){return new Y(await super._call(e))}},qv=class extends Uv{async _call(e){return new Lm(await super._call(e))}},Jv=class extends Uv{async _call(e){return new zm(await super._call(e))}},Yv=class extends Z{},Xv=class extends Yv{},Zv=class extends Yv{},Qv=class extends Z{},$v=class extends Qv{},ey=class extends Qv{async _call(e){return new Rm(await super._call(e))}},ty=class extends Qv{async _call(e){return new Y(await super._call(e))}},ny=class extends Qv{async _call(e){return new Lm(await super._call(e))}},ry=class extends Z{},iy=class extends ry{},ay=class extends ry{async _call(e){return new Rm(await super._call(e))}},oy=class extends ry{async _call(e){return new Y(await super._call(e))}},sy=class extends ry{async _call(e){return new Lm(await super._call(e))}},cy=class extends Z{},ly=class extends cy{},uy=class extends cy{},dy=class extends Z{},fy=class extends dy{},py=class extends dy{},my=class extends Z{},hy=class extends my{},gy=class extends my{},_y=class extends Z{},vy=class extends _y{},yy=class extends _y{async _call(e){return new Y(await super._call(e))}},by=class extends Z{forward_params=[`input_ids`,`inputs_embeds`,`attention_mask`,`pixel_values`,`encoder_outputs`,`decoder_input_ids`,`decoder_inputs_embeds`,`decoder_attention_mask`,`past_key_values`];main_input_name=`inputs_embeds`},xy=class extends by{_merge_input_ids_with_image_features({inputs_embeds:e,image_features:t,input_ids:n,attention_mask:r}){return{inputs_embeds:fl([t,e],1),attention_mask:fl([xl(t.dims.slice(0,2)),r],1)}}async _prepare_inputs_embeds({input_ids:e,pixel_values:t,inputs_embeds:n,attention_mask:r}){if(!e&&!t)throw Error("Either `input_ids` or `pixel_values` should be provided.");let i,a;return e&&(i=await this.encode_text({input_ids:e})),t&&(a=await this.encode_image({pixel_values:t})),i&&a?{inputs_embeds:n,attention_mask:r}=this._merge_input_ids_with_image_features({inputs_embeds:i,image_features:a,input_ids:e,attention_mask:r}):n=i||a,{inputs_embeds:n,attention_mask:r}}async forward({input_ids:e,pixel_values:t,attention_mask:n,decoder_input_ids:r,decoder_attention_mask:i,encoder_outputs:a,past_key_values:o,inputs_embeds:s,decoder_inputs_embeds:c}){if(s||({inputs_embeds:s,attention_mask:n}=await this._prepare_inputs_embeds({input_ids:e,pixel_values:t,inputs_embeds:s,attention_mask:n})),!a){let{last_hidden_state:e}=await Oh(this,{inputs_embeds:s,attention_mask:n});a=e}if(!c){if(!r)throw Error("Either `decoder_input_ids` or `decoder_inputs_embeds` should be provided.");c=await this.encode_text({input_ids:r})}let l={inputs_embeds:c,attention_mask:i,encoder_attention_mask:n,encoder_hidden_states:a,past_key_values:o};return await Ph(this,l,!0)}},Sy=class extends Z{},Cy=class extends Sy{},wy=class extends Sy{},Ty=class extends Z{},Ey=class extends Ty{},Dy=class extends Ty{},Oy=class extends Z{forward_params=[`input_ids`,`attention_mask`,`pixel_values`,`position_ids`,`past_key_values`]},ky=class extends Oy{_merge_input_ids_with_image_features(e){let t=e.image_features.dims.at(-1),n=e.image_features.view(-1,t);return Wh({image_token_id:this.config.image_token_index??this.config.image_token_id,...e,image_features:n})}},Ay=class extends ky{},jy=class extends ky{},My=class extends Z{},Ny=class extends My{},Py=class extends ky{},Fy=class extends Py{},Iy=class extends Z{forward_params=[`input_ids`,`attention_mask`,`inputs_embeds`,`per_layer_inputs`,`position_ids`,`pixel_values`,`input_features`,`input_features_mask`,`past_key_values`]},Ly=class extends Iy{async forward({input_ids:e=null,attention_mask:t=null,pixel_values:n=null,input_features:r=null,input_features_mask:i=null,position_ids:a=null,inputs_embeds:o=null,per_layer_inputs:s=null,past_key_values:c=null,generation_config:l=null,logits_processor:u=null,...d}){if((!o||!s)&&({inputs_embeds:o,per_layer_inputs:s}=await J(this.sessions.embed_tokens,{input_ids:e}),e.dims[1]!==1)){if(n){let{image_features:r}=await this._encode_vision({pixel_values:n,...d});({inputs_embeds:o,attention_mask:t}=this._merge_input_ids_with_image_features({image_features:r,inputs_embeds:o,input_ids:e,attention_mask:t}))}if(r){let{audio_features:n}=await J(this.sessions.audio_encoder,{input_features:r,input_features_mask:i});({inputs_embeds:o,attention_mask:t}=this._merge_input_ids_with_audio_features({audio_features:n,inputs_embeds:o,input_ids:e,attention_mask:t}))}}return await Ph(this,{inputs_embeds:o,per_layer_inputs:s,past_key_values:c,attention_mask:t,position_ids:a,generation_config:l,logits_processor:u},!0)}_encode_vision(e){return J(this.sessions.vision_encoder,{pixel_values:e.pixel_values})}_merge_input_ids_with_image_features(e){let t=e.image_features.dims.at(-1),n=e.image_features.view(-1,t);return Wh({image_token_id:this.config.image_token_id,...e,image_features:n})}_merge_input_ids_with_audio_features(e){let t=e.audio_features.dims.at(-1),n=e.audio_features.view(-1,t);return Gh({audio_token_id:this.config.audio_token_id,...e,audio_features:n})}},Ry=class extends Ly{},zy=class extends Ly{forward_params=[`input_ids`,`attention_mask`,`inputs_embeds`,`per_layer_inputs`,`position_ids`,`pixel_values`,`image_position_ids`,`input_features`,`input_features_mask`,`past_key_values`];_encode_vision(e){return J(this.sessions.vision_encoder,{pixel_values:e.pixel_values,pixel_position_ids:e.image_position_ids})}},By=class extends zy{},Vy=class extends Z{},Hy=class extends Vy{},Uy=class extends Vy{},Wy=class extends Z{},Gy=class extends Wy{},Ky=class extends Wy{},qy=class extends Z{forward_params=[`input_ids`,`attention_mask`,`position_ids`,`past_key_values`,`pixel_values`,`image_grid_thw`]},Jy=class extends qy{image_grid_thw_name=`grid_thw`;_get_text_only_rope_index(e,t){if(t){let{data:e,dims:n}=Rh(t),r=BigInt64Array.from({length:3*e.length},(t,n)=>e[n%e.length]),i=Array.from({length:n[0]},(t,r)=>lc(e.subarray(n[1]*r,n[1]*(r+1)))[0]+1n+BigInt(n[1]));return[new U(`int64`,r,[3,...n]),new U(`int64`,i,[i.length,1])]}else{let[t,n]=e.dims;return[new U(`int64`,BigInt64Array.from({length:3*t*n},(e,r)=>BigInt(Math.floor(r%n/t))),[3,...e.dims]),Cl([t,1])]}}_reorder_and_write_positions(e,t,n,r){let i=e.reduce((e,t)=>e+t.length,0),a=Array(i),o=0;for(let t=0;t<3;++t)for(let n of e){let e=n.length/3;for(let r=t*e;r<(t+1)*e;++r)a[o++]=n[r]}let s=0;for(let e=0;e<t.length;++e)if(t[e]==1){for(let t=0;t<3;++t)n[t][r][e]=a[t*i/3+s];++s}return a}_get_multimodal_rope_positions({filtered_ids:e,image_grid_thw_list:t,video_grid_thw_list:n,spatial_merge_size:r,state:i}){let{image_token_id:a,video_token_id:o,vision_start_token_id:s}=this.config,c=e,l=c.reduce((e,t,n)=>(t==s&&e.push(n),e),[]).map(e=>c[e+1]),u=l.filter(e=>e==a).length,d=l.filter(e=>e==o).length,f=[],p=0,m=u,h=d;for(let e=0;e<l.length;++e){let e=c.findIndex((e,t)=>t>p&&e==a),s=c.findIndex((e,t)=>t>p&&e==o),l=m>0&&e!==-1?e:c.length+1,u=h>0&&s!==-1?s:c.length+1,d,g,_,v;l<u?([g,_,v]=t[i.image_index],++i.image_index,--m,d=l):([g,_,v]=n[i.video_index],++i.video_index,--h,d=u);let[y,b,x]=[Number(g),Math.floor(Number(_)/r),Math.floor(Number(v)/r)],ee=d-p,S=f.length>0?lc(f.at(-1))[0]+1:0;f.push(Array.from({length:3*ee},(e,t)=>S+t%ee));let te=ee+S,C=y*b*x,w=Array.from({length:C},(e,t)=>te+Math.floor(t/(b*x))),ne=Array.from({length:C},(e,t)=>te+Math.floor(t/x)%b),re=Array.from({length:C},(e,t)=>te+t%x);f.push([w,ne,re].flat()),p=d+C}if(p<c.length){let e=f.length>0?lc(f.at(-1))[0]+1:0,t=c.length-p;f.push(Array.from({length:3*t},(n,r)=>e+r%t))}return f}get_rope_index(e,t,n,r){let{vision_config:i}=this.config,a=i.spatial_merge_size??2;if(t||n){let i=e.tolist();r||=Sl(e);let o=r.tolist(),s=Array.from({length:3},()=>Array.from({length:e.dims[0]},()=>Array.from({length:e.dims[1]},()=>0))),c=t?t.tolist():[],l=n?n.tolist():[],u={image_index:0,video_index:0},d=[];for(let e=0;e<i.length;++e){let t=i[e].filter((t,n)=>o[e][n]==1),n=this._get_multimodal_rope_positions({filtered_ids:t,image_grid_thw_list:c,video_grid_thw_list:l,spatial_merge_size:a,state:u}),r=this._reorder_and_write_positions(n,o[e],s,e);d.push(lc(r)[0]+1-i[e].length)}return[new U(`int64`,s.flat(1/0),[3,e.dims[0],e.dims[1]]),new U(`int64`,d,[d.length,1])]}else return this._get_text_only_rope_index(e,r)}async encode_image({pixel_values:e,image_grid_thw:t}){return(await J(this.sessions.vision_encoder,{pixel_values:e,[this.image_grid_thw_name]:t})).image_features}_merge_input_ids_with_image_features(e){return Wh({image_token_id:this.config.image_token_id,...e})}prepare_inputs_for_generation(e,t,n){if(!t.attention_mask||t.position_ids||!(this.sessions.decoder_model_merged??this.sessions.model).inputNames.includes(`position_ids`))return t;if(!t.past_key_values)[t.position_ids,t.rope_deltas]=this.get_rope_index(t.input_ids,t.image_grid_thw,t.video_grid_thw,t.attention_mask);else{t.pixel_values=null;let e=t.past_key_values.get_seq_length();if(e<t.input_ids.dims[1]){let[n,r]=this.get_rope_index(t.input_ids,t.image_grid_thw,t.video_grid_thw,t.attention_mask);t.rope_deltas=r,t.position_ids=n.slice(null,null,[e,null]),t.input_ids=t.input_ids.slice(null,[e,null])}else{t.rope_deltas||([,t.rope_deltas]=this.get_rope_index(t.input_ids,t.image_grid_thw,t.video_grid_thw,t.attention_mask));let n=BigInt(e),r=t.rope_deltas.map(e=>n+e);t.position_ids=pl([r,r,r],0)}}return t}},Yy=class extends Jy{},Xy=class extends Jy{image_grid_thw_name=`image_grid_thw`},Zy=class extends Yy{image_grid_thw_name=`image_grid_thw`},Qy=class extends Xy{get_vision_position_ids(e,t,n,r){let i=Math.floor(t[0]/n),a=Math.floor(t[1]/r),o=Math.floor(t[2]/r),s=a*o*i,c=Array.from({length:s},()=>e),l=Array.from({length:s},(t,n)=>e+Math.floor(n/(o*i))),u=Array.from({length:s},(t,n)=>e+n%o);return[...c,...l,...u]}_get_multimodal_rope_positions({filtered_ids:e,image_grid_thw_list:t,video_grid_thw_list:n,spatial_merge_size:r,state:i}){let{image_token_id:a}=this.config,o=[],s=0,c=+(e[0]==a);for(let t=1;t<=e.length;++t){let n=t<e.length?+(e[t]==a):-1;n!==c&&(o.push([c,s,t]),s=t,c=n)}let l=0,u=[];for(let[e,n,a]of o)if(e===0){let e=a-n;u.push(Array.from({length:3*e},(t,n)=>l+n%e)),l+=e}else{let e=t[i.image_index++].map(Number),n=e[0];u.push(this.get_vision_position_ids(l,e,n,r)),l+=Math.max(e[1],e[2])/r}return u}},$y=class extends Z{},eb=class extends $y{},tb=class extends $y{},nb=class extends Z{},rb=class extends nb{},ib=class extends nb{},ab=class extends Z{},ob=class extends ab{},sb=class extends ab{},cb=class extends Z{},lb=class extends cb{},ub=class extends cb{},db=class extends Z{},fb=class extends db{},pb=class extends db{},mb=class extends Z{},hb=class extends mb{},gb=class extends mb{},_b=class extends Z{},vb=class extends _b{},yb=class extends _b{},bb=class extends Z{},xb=class extends bb{},Sb=class extends bb{},Cb=class extends Z{},wb=class extends Cb{},Tb=class extends Cb{},Eb=class extends Z{forward_params=[`input_ids`,`attention_mask`,`position_ids`,`audio_values`,`past_key_values`]},Db=class extends Eb{_merge_input_ids_with_audio_features(e){let t=e.audio_features.dims.at(-1),n=e.audio_features.view(-1,t);return Gh({audio_token_id:this.config.ignore_index??this.config.audio_token_id??this.config.audio_token_index,...e,audio_features:n})}},Ob=class extends Db{forward_params=[`input_ids`,`attention_mask`,`input_features`,`past_key_values`]},kb=class extends Z{},Ab=class extends kb{},jb=class extends Z{},Mb=class extends jb{},Nb=class extends Z{},Pb=class extends Nb{},Fb=class extends Nb{},Ib=class extends Z{},Lb=class extends Ib{},Rb=class extends Ib{async _call(e){return new Y(await super._call(e))}},zb=class extends Z{},Bb=class extends zb{},Vb=class extends zb{async _call(e){return new Bm(await super._call(e))}},Hb=class extends zb{async _call(e){return new Y(await super._call(e))}},Ub=class extends zb{async _call(e){return new Lm(await super._call(e))}},Wb=class extends Z{},Gb=class extends zb{},Kb=class extends zb{async _call(e){return new Bm(await super._call(e))}},qb=class extends zb{async _call(e){return new Y(await super._call(e))}},Jb=class extends Z{},Yb=class extends Jb{},Xb=class extends Jb{},Zb=class extends ky{forward_params=[`input_ids`,`attention_mask`,`pixel_values`,`pixel_attention_mask`,`position_ids`,`past_key_values`]},Qb=class extends Z{},$b=class extends Qb{},ex=class extends Qb{async _call(e){return new Y(await super._call(e))}},tx=class extends Z{},nx=class extends tx{},rx=class extends tx{},ix=class extends Z{},ax=class extends ix{async forward(e){let t=!e.input_ids,n=!e.pixel_values;if(t&&n)throw Error("Either `input_ids` or `pixel_values` should be provided.");if(t&&(e.input_ids=xl([e.pixel_values.dims[0],1])),n){let{image_size:t}=this.config.vision_config;e.pixel_values=yl([0,3,t,t],0)}let{text_embeddings:r,image_embeddings:i,l2norm_text_embeddings:a,l2norm_image_embeddings:o}=await super.forward(e),s={};return t||(s.text_embeddings=r,s.l2norm_text_embeddings=a),n||(s.image_embeddings=i,s.l2norm_image_embeddings=o),s}},ox=class extends ix{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`text_model`})}},sx=class extends ix{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`vision_model`})}},cx=class extends Z{},lx=class extends cx{},ux=class extends cx{},dx=class extends ky{},fx=class extends Z{},px=class extends fx{},mx=class extends fx{},hx=class extends ky{forward_params=[`input_ids`,`attention_mask`,`pixel_values`,`pixel_attention_mask`,`spatial_shapes`,`position_ids`,`past_key_values`]},gx=class extends Z{},_x=class extends gx{},vx=class extends gx{},yx=class extends Z{},bx=class extends yx{},xx=class extends Z{},Sx=class extends xx{},Cx=class extends xx{},wx=class extends Z{},Tx=class extends wx{},Ex=class extends wx{},Dx=class extends Z{},Ox=class extends Dx{},kx=class extends Dx{},Ax=class extends Z{},jx=class extends Ax{},Mx=class extends Ax{},Nx=class extends Z{},Px=class extends Nx{},Fx=class extends Nx{},Ix=class extends Nx{async _call(e){return new Y(await super._call(e))}},Lx=class extends Nx{},Rx=class extends Z{},zx=class extends Rx{},Bx=class extends Z{},Vx=class extends Bx{},Hx=class extends Im{constructor({char_logits:e,bpe_logits:t,wp_logits:n}){super(),this.char_logits=e,this.bpe_logits=t,this.wp_logits=n}get logits(){return[this.char_logits,this.bpe_logits,this.wp_logits]}},Ux=class extends Z{},Wx=class extends Ux{async _call(e){return new Hx(await super._call(e))}},Gx=class extends Im{constructor({audio_codes:e}){super(),this.audio_codes=e}},Kx=class extends Im{constructor({audio_values:e}){super(),this.audio_values=e}},qx=class extends Z{main_input_name=`input_values`;forward_params=[`input_values`]},Jx=class extends qx{async encode(e){return new Gx(await J(this.sessions.encoder_model,e))}async decode(e){return new Kx(await J(this.sessions.decoder_model,e))}},Yx=class extends qx{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`encoder_model`})}},Xx=class extends qx{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`decoder_model`})}},Zx=class extends Z{},Qx=class extends Zx{},$x=class extends Zx{},eS=class extends Z{},tS=class extends eS{},nS=class extends eS{},rS=class extends Z{},iS=class extends rS{},aS=class extends rS{async _call(e){return new Rm(await super._call(e))}},oS=class extends rS{async _call(e){return new Y(await super._call(e))}},sS=class extends rS{async _call(e){return new zm(await super._call(e))}},cS=class extends Z{},lS=class extends cS{},uS=class extends cS{},dS=class extends Z{},fS=class extends dS{},pS=class extends dS{async _call(e){return new Y(await super._call(e))}},mS=class extends dS{},hS=class extends Z{},gS=class extends hS{},_S=class extends hS{async _call(e){return new Y(await super._call(e))}},vS=class extends hS{},yS=class extends Z{},bS=class extends yS{},xS=class extends yS{async _call(e){return new Y(await super._call(e))}},SS=class extends yS{},CS=class extends Z{},wS=class extends CS{},TS=class extends CS{async _call(e){return new Y(await super._call(e))}},ES=class extends CS{},DS=class extends Z{},OS=class extends DS{},kS=class extends DS{async _call(e){return new Y(await super._call(e))}},AS=class extends Z{},jS=class extends AS{},MS=class extends AS{async _call(e){return new Y(await super._call(e))}},NS=class extends Z{},PS=class extends NS{},FS=class extends NS{async _call(e){return new Rm(await super._call(e))}},IS=class extends NS{async _call(e){return new Y(await super._call(e))}},LS=class extends NS{async _call(e){return new Lm(await super._call(e))}},RS=class extends Z{},zS=class extends RS{},BS=class extends RS{},VS=class extends Z{requires_attention_mask=!1;main_input_name=`input_values`;forward_params=[`input_values`,`decoder_input_ids`,`past_key_values`]},HS=class extends VS{},US=class extends VS{},WS=class extends Z{},GS=class extends WS{},KS=class extends WS{async _call(e){return new Rm(await super._call(e))}},qS=class extends WS{async _call(e){return new Y(await super._call(e))}},JS=class extends WS{async _call(e){return new Lm(await super._call(e))}},YS=class extends WS{async _call(e){return new zm(await super._call(e))}},XS=class extends Z{},ZS=class extends XS{},QS=class extends XS{},$S=class extends Z{},eC=class extends $S{},tC=class extends $S{},nC=class extends Z{},rC=class extends nC{forward_params=[`input_ids`,`pixel_values`,`images_seq_mask`,`images_emb_mask`,`attention_mask`,`position_ids`,`past_key_values`];constructor(...e){super(...e),this._generation_mode=`text`}async forward(e){let t=this._generation_mode??`text`,n;if(t===`text`||!e.past_key_values){let t=this.sessions.prepare_inputs_embeds;n=await J(t,di(e,t.inputNames))}else{let t=this.sessions.gen_img_embeds;n=await J(t,di({image_ids:e.input_ids},t.inputNames))}let r={...e,...n},i=await Ph(this,r),a=this.sessions[t===`text`?`lm_head`:`gen_head`];if(!a)throw Error(`Unable to find "${a}" generation head`);let o=await J(a,di(i,a.inputNames));return{...n,...i,...o}}prepare_inputs_for_generation(e,t,n){let r=!!t.past_key_values;return n.guidance_scale!==null&&n.guidance_scale>1&&(r?t.input_ids=fl([t.input_ids,t.input_ids],0):(t.input_ids=fl([t.input_ids,bl(t.input_ids,BigInt(n.pad_token_id))],0),t.attention_mask=fl([t.attention_mask,bl(t.attention_mask,0n)],0))),(r||!t.pixel_values)&&(t.pixel_values=yl([0,0,3,384,384],1)),r&&(t.images_seq_mask=new U(`bool`,[,].fill(!0).fill(!1,0,1),[1,1]),t.images_emb_mask=new U(`bool`,[].fill(!1),[1,1,0])),t}async generate(e){return this._generation_mode=`text`,super.generate(e)}async generate_images(e){this._generation_mode=`image`;let t=(e.inputs??e[this.main_input_name]).dims[1],n=(await super.generate(e)).slice(null,[t,null]),r=this.sessions.image_decode,{decoded_image:i}=await J(r,{generated_tokens:n}),a=i.add_(1).mul_(255/2).clamp_(0,255).to(`uint8`),o=[];for(let e of a){let t=Ud.fromTensor(e);o.push(t)}return o}},iC=class extends Z{},aC=class extends iC{},oC=class extends iC{},sC=class extends Z{forward_params=[`input_ids`,`attention_mask`,`encoder_outputs`,`decoder_input_ids`,`decoder_attention_mask`,`past_key_values`];_apply_and_filter_by_delay_pattern_mask(e){let[t,n]=e.dims,r=this.config.decoder.num_codebooks,i=n-r,a=0;for(let t=0;t<e.size;++t){if(e.data[t]==this.config.decoder.pad_token_id)continue;let o=t%n-Math.floor(t/n)%r;o>0&&o<=i&&(e.data[a++]=e.data[t])}let o=Math.floor(t/r),s=a/(o*r);return new U(e.type,e.data.slice(0,a),[o,r,s])}prepare_inputs_for_generation(e,t,n){let r=BigInt(this.config.decoder.pad_token_id),i=structuredClone(e);for(let e=0;e<i.length;++e)for(let t=0;t<i[e].length;++t)e%this.config.decoder.num_codebooks>=t&&(i[e][t]=r);return n.guidance_scale!==null&&n.guidance_scale>1&&(i=i.concat(i)),Vh(this,i,t,n)}async generate(e){let t=await super.generate(e),n=this._apply_and_filter_by_delay_pattern_mask(t).unsqueeze_(0),{audio_values:r}=await J(this.sessions.encodec_decode,{audio_codes:n});return r}},cC=class extends Z{},lC=class extends cC{},uC=class extends cC{},dC=class extends Z{},fC=class extends dC{},pC=class extends dC{},mC=class extends Z{},hC=class extends mC{},gC=class extends mC{async _call(e){return new Rm(await super._call(e))}},_C=class extends mC{async _call(e){return new Y(await super._call(e))}},vC=class extends mC{async _call(e){return new Lm(await super._call(e))}},yC=class extends mC{async _call(e){return new zm(await super._call(e))}},bC=class extends Z{},xC=class extends bC{},SC=class extends Z{},CC=class extends SC{},wC=class extends SC{},TC=class extends Z{},EC=class extends TC{},DC=class extends TC{},OC=class extends Z{},kC=class extends OC{},AC=class extends OC{},jC=class extends Z{},MC=class extends jC{},NC=class extends jC{},PC=class extends Z{},FC=class extends PC{},IC=class extends PC{async _call(e){return new Y(await super._call(e))}},LC=class extends Z{},RC=class extends LC{},zC=class extends LC{},BC=class extends Z{},VC=class extends BC{},HC=class extends BC{},UC=class extends Z{},WC=class extends UC{},GC=class extends UC{},KC=class extends Z{},qC=class extends KC{},JC=class extends KC{},YC=class extends ky{},XC=class extends Z{},ZC=class extends XC{async _call(e){return new Bm(await super._call(e))}},QC=class extends Z{},$C=class extends QC{},ew=class extends QC{},tw=class extends Z{},nw=class extends tw{},rw=class extends tw{},iw=class extends Z{},aw=class extends iw{},ow=class extends iw{},sw=class extends Z{},cw=class extends sw{},lw=class extends sw{},uw=class extends Z{forward_params=[`input_ids`,`inputs_embeds`,`attention_mask`,`position_ids`,`pixel_values`,`image_sizes`,`past_key_values`]},dw=class extends uw{async forward({input_ids:e=null,attention_mask:t=null,pixel_values:n=null,image_sizes:r=null,position_ids:i=null,inputs_embeds:a=null,past_key_values:o=null,generation_config:s=null,logits_processor:c=null,...l}){if(!a){let t;if(n&&e.dims[1]!==1){if(!r)throw Error("`image_sizes` must be provided when `pixel_values` is provided.");({image_features:t}=await J(this.sessions.vision_encoder,{pixel_values:n,image_sizes:r}))}else{let e=this.config.normalized_config.hidden_size;t=new U(`float32`,[],[0,e])}({inputs_embeds:a}=await J(this.sessions.prepare_inputs_embeds,{input_ids:e,image_features:t}))}return await Ph(this,{inputs_embeds:a,past_key_values:o,attention_mask:t,position_ids:i,generation_config:s,logits_processor:c},!1)}},fw=class extends Z{},pw=class extends fw{},mw=class extends fw{async _call(e){return new Y(await super._call(e))}},hw=class extends Z{},gw=class extends hw{},_w=class extends hw{async _call(e){return new Lm(await super._call(e))}},vw=class extends Z{},yw=class extends vw{},bw=class extends vw{},xw=class extends Z{},Sw=class extends xw{},Cw=class extends xw{},ww=class extends Z{},Tw=class extends ww{},Ew=class extends ww{},Dw=class extends Z{},Ow=class extends Dw{},kw=class extends Dw{},Aw=class extends Z{},jw=class extends Aw{},Mw=class extends Aw{},Nw=class extends Xy{},Pw=class extends Zy{},Fw=class extends Nw{},Iw=class extends Pw{},Lw=class extends Nw{},Rw=class extends Lw{},zw=class extends Lw{},Bw=class extends Rw{},Vw=class extends Z{},Hw=class extends Vw{},Uw=class extends Vw{async _call(e){return new Y(await super._call(e))}},Ww=class extends Z{},Gw=class extends Ww{},Kw=class extends Ww{async _call(e){return new qw(await super._call(e))}},qw=class extends N_{},Jw=class extends Z{},Yw=class extends Jw{},Xw=class extends Jw{async _call(e){return new Rm(await super._call(e))}},Zw=class extends Jw{async _call(e){return new Y(await super._call(e))}},Qw=class extends Jw{async _call(e){return new Lm(await super._call(e))}},$w=class extends Jw{async _call(e){return new zm(await super._call(e))}},eT=class extends Z{},tT=class extends eT{},nT=class extends eT{async _call(e){return new Rm(await super._call(e))}},rT=class extends eT{async _call(e){return new Y(await super._call(e))}},iT=class extends eT{async _call(e){return new Lm(await super._call(e))}},aT=class extends eT{async _call(e){return new zm(await super._call(e))}},oT=class extends Z{},sT=class extends oT{},cT=class extends oT{async _call(e){return new lT(await super._call(e))}},lT=class extends N_{},uT=class extends Im{constructor({iou_scores:e,pred_masks:t}){super(),this.iou_scores=e,this.pred_masks=t}},dT=class extends Z{},fT=class extends dT{async get_image_embeddings({pixel_values:e}){return await Oh(this,{pixel_values:e})}async forward(e){e=!e.image_embeddings||!e.image_positional_embeddings?{...e,...await this.get_image_embeddings(e)}:{...e},e.input_labels??=xl(e.input_points.dims.slice(0,-1));let t={image_embeddings:e.image_embeddings,image_positional_embeddings:e.image_positional_embeddings};return e.input_points&&(t.input_points=e.input_points),e.input_labels&&(t.input_labels=e.input_labels),e.input_boxes&&(t.input_boxes=e.input_boxes),await J(this.sessions.prompt_encoder_mask_decoder,t)}async _call(e){return new uT(await super._call(e))}},pT=class extends Im{constructor({iou_scores:e,pred_masks:t,object_score_logits:n}){super(),this.iou_scores=e,this.pred_masks=t,this.object_score_logits=n}},mT=class extends Z{},hT=class extends mT{async get_image_embeddings({pixel_values:e}){return await Oh(this,{pixel_values:e})}async forward(e){let{num_feature_levels:t}=this.config.vision_config;if(e=Array.from({length:t},(e,t)=>`image_embeddings.${t}`).some(t=>!e[t])?{...e,...await this.get_image_embeddings(e)}:{...e},e.input_points){if(e.input_boxes&&e.input_boxes.dims[1]!==1)throw Error("When both `input_points` and `input_boxes` are provided, the number of boxes per image must be 1.");let t=e.input_points.dims;e.input_labels??=xl(t.slice(0,-1)),e.input_boxes??=yl([t[0],0,4],0)}else if(e.input_boxes){let t=e.input_boxes.dims;e.input_labels=yl([t[0],t[1],0],-1n),e.input_points=yl([t[0],1,0,2],0)}else throw Error("At least one of `input_points` or `input_boxes` must be provided.");let n=this.sessions.prompt_encoder_mask_decoder;return await J(n,di(e,n.inputNames))}async _call(e){return new pT(await super._call(e))}},gT=class extends hT{},_T=class extends hT{},vT=class extends Z{},yT=class extends vT{},bT=class extends vT{},xT=class extends vT{},ST=class extends Z{},CT=class extends ST{},wT=class extends ST{},TT=class extends ST{},ET=class extends Z{},DT=class extends ET{},OT=class extends ET{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`text_model`})}},kT=class extends Zg{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`vision_model`})}},AT=class extends Z{},jT=class extends AT{},MT=class extends AT{},NT=class extends Zb{},PT=class extends Z{main_input_name=`input_values`;forward_params=[`input_values`]},FT=class extends PT{async encode(e){return await J(this.sessions.encoder_model,e)}async decode(e){return await J(this.sessions.decoder_model,e)}},IT=class extends PT{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`encoder_model`})}},LT=class extends PT{static async from_pretrained(e,t={}){return super.from_pretrained(e,{...t,model_file_name:t.model_file_name??`decoder_model`})}},RT=class extends Z{},zT=class extends RT{},BT=class extends RT{},VT=class extends Z{},HT=class extends VT{},UT=class extends VT{},WT=class extends VT{async generate_speech(e,t,{threshold:n=.5,minlenratio:r=0,maxlenratio:i=20,vocoder:a=null}={}){let o={input_ids:e},{encoder_outputs:s,encoder_attention_mask:c}=await Oh(this,o),l=s.dims[1]/this.config.reduction_factor,u=Math.floor(l*i),d=Math.floor(l*r),f=this.config.num_mel_bins,p=[],m=null,h=null,g=0;for(;;){++g;let e=xh(!!h),r;r=h?h.output_sequence_out:new U(`float32`,new Float32Array(f),[1,1,f]);let i={use_cache_branch:e,output_sequence:r,encoder_attention_mask:c,speaker_embeddings:t,encoder_hidden_states:s};Nh(this,i,m),h=await J(this.sessions.decoder_model_merged,i),m=Ah(h,m);let{prob:a,spectrum:o}=h;if(p.push(o),g>=d&&(Array.from(a.data).filter(e=>e>=n).length>0||g>=u))break}let _=fl(p),{waveform:v}=await J(a.sessions.model,{spectrogram:_});return{spectrogram:_,waveform:v}}},GT=class extends Z{main_input_name=`spectrogram`},KT=class extends Z{},qT=class extends KT{},JT=class extends KT{async _call(e){return new Rm(await super._call(e))}},YT=class extends KT{async _call(e){return new Y(await super._call(e))}},XT=class extends KT{async _call(e){return new zm(await super._call(e))}},ZT=class extends Z{},QT=class extends ZT{},$T=class extends ZT{},eE=class extends Z{},tE=class extends eE{},nE=class extends eE{},rE=class extends Z{},iE=class extends rE{},aE=class extends Z{},oE=class extends aE{async generate_speech({input_ids:e,attention_mask:t,style:n,num_inference_steps:r=5,speed:i=1.05}){let{sampling_rate:a,chunk_compress_factor:o,base_chunk_size:s,latent_dim:c}=this.config,{last_hidden_state:l,durations:u}=await J(this.sessions.text_encoder,{input_ids:e,attention_mask:t,style:n}),d=u.div(i).mul_(a),f=s*o,p=d.data,m=Int32Array.from(p,e=>Math.ceil(e/f)),h=Math.max(...m),g=e.dims[0],_=new BigInt64Array(g*h);for(let e=0;e<g;++e)_.fill(1n,e*h,e*h+m[e]);let v=new U(`int64`,_,[g,h]),y=c*o,b=y*h,x=Tl([g,y,h]),ee=x.data;for(let e=0;e<g;++e)if(m[e]!==h)for(let t=0;t<y;++t)ee.fill(0,e*b+t*h+m[e],e*b+(t+1)*h);let S=yl([g],r);for(let e=0;e<r;++e){let r=yl([g],e);({denoised_latents:x}=await J(this.sessions.latent_denoiser,{style:n,noisy_latents:x,latent_mask:v,encoder_outputs:l,attention_mask:t,timestep:r,num_inference_steps:S}))}let{waveform:te}=await J(this.sessions.voice_decoder,{latents:x});return{waveform:te,durations:d}}},sE=class extends Z{},cE=class extends sE{},lE=class extends sE{async _call(e){return new Y(await super._call(e))}},uE=class extends sE{},dE=class extends Z{},fE=class extends dE{},pE=class extends dE{},mE=class extends Z{forward_params=[`input_ids`,`attention_mask`,`encoder_outputs`,`decoder_input_ids`,`decoder_attention_mask`,`past_key_values`]},hE=class extends mE{},gE=class extends mE{},_E=class extends Z{},vE=class extends _E{},yE=class extends _E{async _call(e){return new bE(await super._call(e))}},bE=class extends _v{},xE=class extends Z{},SE=class extends xE{},CE=class extends Z{},wE=class extends CE{},TE=class extends CE{async _call(e){return new Bm(await super._call(e))}},EE=class extends CE{async _call(e){return new Y(await super._call(e))}},DE=class extends Z{},OE=class extends DE{},kE=class extends DE{async _call(e){return new Bm(await super._call(e))}},AE=class extends DE{async _call(e){return new Y(await super._call(e))}},jE=class extends DE{async _call(e){return new Lm(await super._call(e))}},ME=class extends Z{},NE=class extends ME{},PE=class extends ME{},FE=class extends Z{main_input_name=`pixel_values`;forward_params=[`pixel_values`,`decoder_input_ids`,`encoder_hidden_states`,`past_key_values`]},IE=class extends Z{},LE=class extends IE{},RE=class extends IE{async _call(e){return new Y(await super._call(e))}},zE=class extends Z{},BE=class extends zE{},VE=class extends Z{},HE=class extends VE{},UE=class extends VE{async _call(e){return new Y(await super._call(e))}},WE=class extends Z{},GE=class extends WE{async _call(e){return new Vm(await super._call(e))}},KE=class extends Z{},qE=class extends KE{},JE=class extends Im{constructor({waveform:e,spectrogram:t}){super(),this.waveform=e,this.spectrogram=t}},YE=class extends Z{},XE=class extends YE{async _call(e){return new JE(await super._call(e))}},ZE=class extends Db{},QE=2,$E=1,eD=new WeakMap;function tD(e,t){let{text_config:n,audio_config:r}=e.config,i=e.sessions.audio_encoder,{num_mel_bins:a,hidden_size:o}=r,s=a+o,c=new fh,l=wm(r),u={batch_size:1},d=`float32`;for(let e of i.inputMetadata){if(e.name===`past_padding_cache`){d=e.type;continue}if(!l.has(e.name))continue;let t=Mh(e.shape,u),n=t.reduce((e,t)=>e*t,1),r=$c[e.type];c[e.name]=new U(e.type,new r(n),t)}let f=$c[d],p=new U(d,new f(s*QE),[1,s,QE]),m=t[Symbol.asyncIterator]?.()??t[Symbol.iterator]?.();if(!m)throw Error(`input_features must be iterable or async iterable`);return{encoder_session:i,enc_kv_cache:c,enc_padding_cache:p,enc_past_seq_len:0,audio_embed_queue:[],audio_embed_total_tokens:0,audio_queue_offset:0,audio_consumed:0,stream_exhausted:!1,chunks_iter:m,text_hidden_size:n.hidden_size}}async function nD(e,t){let n=t.dims[2],r=Math.floor(($E+n-3)/2)+1,i=new U(`int64`,BigInt64Array.from({length:r},(t,n)=>BigInt(e.enc_past_seq_len+n)),[1,r]),a=e.enc_past_seq_len+r,o=xl([1,a]),{audio_embeds:s,present_padding_cache:c,...l}=await J(e.encoder_session,{input_features:t,attention_mask:o,position_ids:i,past_padding_cache:e.enc_padding_cache,...e.enc_kv_cache});e.enc_padding_cache.location===`gpu-buffer`&&e.enc_padding_cache.dispose(),e.enc_padding_cache=c;for(let t in l)if(t.startsWith(`present.`)){let n=t.replace(`present`,`past_key_values`),r=e.enc_kv_cache[n];r?.location===`gpu-buffer`&&r.dispose(),e.enc_kv_cache[n]=l[t]}return e.enc_past_seq_len=a,s}async function rD(e,t){for(;e.audio_embed_total_tokens<t&&!e.stream_exhausted;){let t=await e.chunks_iter.next();if(t.done){e.stream_exhausted=!0;break}let n=await nD(e,t.value);e.audio_embed_queue.push({data:n.data,tokens:n.dims[1]}),e.audio_embed_total_tokens+=n.dims[1]}}function iD(e,t,n){if(e.audio_embed_queue.length===0)return;let r=t.data,i=0,a=n;for(;a>0&&e.audio_embed_queue.length>0;){let t=e.audio_embed_queue[0],n=t.tokens-e.audio_queue_offset,o=Math.min(a,n),s=e.audio_queue_offset*e.text_hidden_size;for(let n=0;n<o*e.text_hidden_size;++n)r[i*e.text_hidden_size+n]+=t.data[s+n];i+=o,a-=o,e.audio_queue_offset+=o,e.audio_queue_offset>=t.tokens&&(e.audio_embed_queue.shift(),e.audio_queue_offset=0)}e.audio_consumed+=n-a}var aD=class extends ih{constructor(e){super(),this._s=e}_call(e){let t=this._s.stream_exhausted&&this._s.audio_embed_queue.length===0;return e.map(()=>t)}},oD=class extends Z{forward_params=[`input_ids`,`attention_mask`,`position_ids`,`past_key_values`]},sD=class extends oD{async forward({input_ids:e,past_key_values:t,...n}){let r=e.dims[1],i=eD.get(this);i&&await rD(i,i.audio_consumed+r);let{inputs_embeds:a}=await J(this.sessions.embed_tokens,{input_ids:e});i&&iD(i,a,r);let o={inputs_embeds:a,...n};Nh(this,o,t);let s=this.sessions.decoder_model_merged;return await J(s,di(o,s.inputNames))}async generate({input_features:e,stopping_criteria:t,...n}){if(!e)throw Error(`input_features (generator/iterable) must be provided`);let r=tD(this,e);eD.set(this,r);let i=new ah;i.push(new aD(r)),t&&i.extend(t);try{return await super.generate({...n,stopping_criteria:i})}finally{r.enc_kv_cache.dispose(),eD.delete(this)}}},cD=class extends Z{},lD=class extends cD{},uD=class extends cD{async _call(e){return new Bm(await super._call(e))}},dD=class extends cD{async _call(e){return new Y(await super._call(e))}},fD=class extends Im{constructor({logits:e,embeddings:t}){super(),this.logits=e,this.embeddings=t}},pD=class extends Z{},mD=class extends pD{},hD=class extends pD{async _call(e){return new Bm(await super._call(e))}},gD=class extends pD{async _call(e){return new Y(await super._call(e))}},_D=class extends pD{async _call(e){return new fD(await super._call(e))}},vD=class extends pD{async _call(e){return new Lm(await super._call(e))}},yD=class extends Z{},bD=class extends yD{},xD=class extends rh{return_timestamps=null;return_token_timestamps=null;num_frames=null;alignment_heads=null;task=null;language=null;no_timestamps_token_id=null;prompt_ids=null;is_multilingual=null;lang_to_id=null;task_to_id=null;max_initial_timestamp_index=1},SD=class extends Z{requires_attention_mask=!1;main_input_name=`input_features`;forward_params=[`input_features`,`attention_mask`,`decoder_input_ids`,`decoder_attention_mask`,`past_key_values`]},CD=class extends SD{},wD=class extends SD{_prepare_generation_config(e,t){return super._prepare_generation_config(e,t,xD)}_retrieve_init_tokens(e){let t=[e.decoder_start_token_id],n=e.language,r=e.task;if(e.is_multilingual){n||=(N.warn(`No language specified - defaulting to English (en).`),`en`);let i=`<|${ku(n)}|>`;t.push(e.lang_to_id[i]),t.push(e.task_to_id[r??`transcribe`])}else if(n||r)throw Error("Cannot specify `task` or `language` for an English-only model. If the model is intended to be multilingual, pass `is_multilingual=true` to generate, or update the generation config.");return!e.return_timestamps&&e.no_timestamps_token_id&&t.at(-1)!==e.no_timestamps_token_id?t.push(e.no_timestamps_token_id):e.return_timestamps&&t.at(-1)===e.no_timestamps_token_id&&(N.warn("<|notimestamps|> prompt token is removed from generation_config since `return_timestamps` is set to `true`."),t.pop()),t.filter(e=>e!=null)}async generate({inputs:e=null,generation_config:t=null,logits_processor:n=null,stopping_criteria:r=null,...i}){t=this._prepare_generation_config(t,i);let a=i.decoder_input_ids instanceof U?kl(i.decoder_input_ids):i.decoder_input_ids??this._retrieve_init_tokens(t);if(t.return_timestamps&&(n??=new Wm,n.push(new Ym(t,a))),t.begin_suppress_tokens&&(n??=new Wm,n.push(new Jm(t.begin_suppress_tokens,a.length))),t.return_token_timestamps){if(!t.alignment_heads)throw Error("Model generation config has no `alignment_heads`, token-level timestamps not available. See https://gist.github.com/hollance/42e32852f24243b748ae6bc1f985b13a on how to add this property to the generation config.");t.task===`translate`&&N.warn(`Token-level timestamps may not be reliable for task 'translate'.`),t.output_attentions=!0,t.return_dict_in_generate=!0}if(t.return_timestamps&&!i.max_new_tokens)return this._generate_with_seek({inputs:e,generation_config:t,logits_processor:n,init_tokens:a,kwargs:i});let o=await super.generate({inputs:e,generation_config:t,logits_processor:n,decoder_input_ids:a,...i});return t.return_token_timestamps&&(o.token_timestamps=this._extract_token_timestamps(o,t.alignment_heads,t.num_frames,.02,a.length)),o}async _generate_with_seek({inputs:e,generation_config:t,logits_processor:n,init_tokens:r,kwargs:i}){let a=t.no_timestamps_token_id+1,o=Array.isArray(t.eos_token_id)?t.eos_token_id[0]:t.eos_token_id,s=t.return_token_timestamps,c=e,l=c.dims[2],u=2*this.config.max_source_positions,d=0,f=[],p=[];for(;d<l;){let e=Math.min(d+u,l),m=c.slice(null,null,[d,e]),h,g=m.dims[2];if(g<u){let e=c.dims[1],t=new Float32Array(e*u),n=m.data;for(let r=0;r<e;++r)t.set(n.subarray(r*g,(r+1)*g),r*u);h=new U(`float32`,t,[1,e,u])}else h=m;if(n)for(let e of n)`begin_index`in e&&(e.begin_index=r.length);let _=await super.generate({inputs:h,generation_config:t,logits_processor:n,decoder_input_ids:r,...i}),v=(s?_.sequences:_)[0].tolist().map(Number).slice(r.length),y;if(s){_.token_timestamps=this._extract_token_timestamps(_,t.alignment_heads,Math.floor((e-d)/2),.02,r.length);let n=d/2*.02;y=_.token_timestamps[0].tolist().slice(r.length).map(e=>e+n)}if(v.length>0&&v.at(-1)===o&&v.pop(),v.length===0)break;let b=v.map(e=>e>=a),x=v.length>=2&&b[v.length-1]&&!b[v.length-2],ee=[];for(let e=0;e<v.length-1;++e)b[e]&&b[e+1]&&ee.push(e+1);let S,te=v.length;if(ee.length>0)if(x)S=e-d;else{let e=ee.at(-1);S=(v[e-1]-a)*2,te=e}else S=e-d;let C=Math.floor(d/2),w=a+1500;for(let e=0;e<te;++e)v[e]>=a&&(v[e]=Math.min(v[e]+C,w));f.push(...v.slice(0,te)),y&&p.push(...y.slice(0,te)),d+=S}f.push(o);let m=[...r,...f];if(s){let e=new U(`int64`,m.map(BigInt),[1,m.length]),t=[...Array(r.length).fill(0),...p,0];return{sequences:e,token_timestamps:new U(`float32`,new Float32Array(t),[1,t.length])}}return new U(`int64`,m.map(BigInt),[1,m.length])}_extract_token_timestamps(e,t,n=null,r=.02,i=0){if(!e.cross_attentions)throw Error("Model outputs must contain cross attentions to extract timestamps. This is most likely because the model was not exported with `output_attentions=True`.");n??N.warn("`num_frames` has not been set, meaning the entire audio will be analyzed. This may lead to inaccurate token-level timestamps for short audios (< 30 seconds).");let a=this.config.median_filter_width;a===void 0&&(N.warn("Model config has no `median_filter_width`, using default value of 7."),a=7);let o=e.cross_attentions,s=Array.from({length:this.config.decoder_layers},(e,t)=>fl(o.map(e=>e[t]),2)),c=pl(t.map(([e,t])=>{if(e>=s.length)throw Error(`Layer index ${e} is out of bounds for cross attentions (length ${s.length}).`);return n?s[e].slice(null,t,null,[0,n]):s[e].slice(null,t)})).transpose(1,0,2,3),[l,u]=hl(c,-2,0,!0),d=c.clone();for(let e=0;e<d.dims[0];++e){let t=d[e];for(let n=0;n<t.dims[0];++n){let r=t[n],i=l[e][n][0].data,o=u[e][n][0].data;for(let e=0;e<r.dims[0];++e){let t=r[e].data;for(let e=0;e<t.length;++e)t[e]=(t[e]-o[e])/i[e];t.set(mc(t,a))}}}let f=[gl(i>0?d.slice(null,null,[i,d.dims[2]],null):d,1)],p=e.sequences.dims,m=new U(`float32`,new Float32Array(p[0]*p[1]),p);for(let e=0;e<p[0];++e){let[t,n]=_c(f[e].neg().squeeze_(0).tolist()),a=ci([1],Array.from({length:t.length-1},(e,n)=>t[n+1]-t[n])).map(e=>!!e),o=[];for(let e=0;e<a.length;++e)a[e]&&o.push(n[e]*r);let s=Array(i).fill(0);s.push(...o),o.length>0&&s.push(o.at(-1)),m[e].data.set(s)}return m}},TD=class extends wD{},ED=class extends Z{},DD=class extends ED{},OD=class extends ED{async _call(e){return new Rm(await super._call(e))}},kD=class extends ED{async _call(e){return new Y(await super._call(e))}},AD=class extends ED{async _call(e){return new Lm(await super._call(e))}},jD=class extends ED{async _call(e){return new zm(await super._call(e))}},MD=class extends Z{},ND=class extends MD{},PD=class extends MD{async _call(e){return new Rm(await super._call(e))}},FD=class extends MD{async _call(e){return new Y(await super._call(e))}},ID=class extends MD{async _call(e){return new Lm(await super._call(e))}},LD=class extends MD{async _call(e){return new zm(await super._call(e))}},RD=class extends Z{},zD=class extends RD{},BD=class extends RD{async _call(e){return new VD(await super._call(e))}},VD=class extends Im{constructor({logits:e,pred_boxes:t}){super(),this.logits=e,this.pred_boxes=t}},HD=class extends Z{},UD=class extends HD{},WD=class extends HD{},GD=new Map([[`bert`,`BertModel`],[`eurobert`,`EuroBertModel`],[`neobert`,`NeoBertModel`],[`modernbert`,`ModernBertModel`],[`nomic_bert`,`NomicBertModel`],[`roformer`,`RoFormerModel`],[`electra`,`ElectraModel`],[`esm`,`EsmModel`],[`convbert`,`ConvBertModel`],[`camembert`,`CamembertModel`],[`deberta`,`DebertaModel`],[`deberta-v2`,`DebertaV2Model`],[`mpnet`,`MPNetModel`],[`albert`,`AlbertModel`],[`distilbert`,`DistilBertModel`],[`roberta`,`RobertaModel`],[`xlm`,`XLMModel`],[`xlm-roberta`,`XLMRobertaModel`],[`clap`,`ClapModel`],[`clip`,`CLIPModel`],[`clipseg`,`CLIPSegModel`],[`chinese_clip`,`ChineseCLIPModel`],[`siglip`,`SiglipModel`],[`jina_clip`,`JinaCLIPModel`],[`mobilebert`,`MobileBertModel`],[`squeezebert`,`SqueezeBertModel`],[`wav2vec2`,`Wav2Vec2Model`],[`wav2vec2-bert`,`Wav2Vec2BertModel`],[`unispeech`,`UniSpeechModel`],[`unispeech-sat`,`UniSpeechSatModel`],[`hubert`,`HubertModel`],[`wavlm`,`WavLMModel`],[`audio-spectrogram-transformer`,`ASTModel`],[`vits`,`VitsModel`],[`pyannote`,`PyAnnoteModel`],[`wespeaker-resnet`,`WeSpeakerResNetModel`],[`detr`,`DetrModel`],[`rt_detr`,`RTDetrModel`],[`rt_detr_v2`,`RTDetrV2Model`],[`rf_detr`,`RFDetrModel`],[`d_fine`,`DFineModel`],[`table-transformer`,`TableTransformerModel`],[`vit`,`ViTModel`],[`ijepa`,`IJepaModel`],[`pvt`,`PvtModel`],[`vit_msn`,`ViTMSNModel`],[`vit_mae`,`ViTMAEModel`],[`groupvit`,`GroupViTModel`],[`fastvit`,`FastViTModel`],[`mobilevit`,`MobileViTModel`],[`mobilevitv2`,`MobileViTV2Model`],[`owlvit`,`OwlViTModel`],[`owlv2`,`Owlv2Model`],[`beit`,`BeitModel`],[`deit`,`DeiTModel`],[`hiera`,`HieraModel`],[`convnext`,`ConvNextModel`],[`convnextv2`,`ConvNextV2Model`],[`dinov2`,`Dinov2Model`],[`dinov2_with_registers`,`Dinov2WithRegistersModel`],[`dinov3_vit`,`DINOv3ViTModel`],[`dinov3_convnext`,`DINOv3ConvNextModel`],[`resnet`,`ResNetModel`],[`swin`,`SwinModel`],[`swin2sr`,`Swin2SRModel`],[`donut-swin`,`DonutSwinModel`],[`yolos`,`YolosModel`],[`dpt`,`DPTModel`],[`glpn`,`GLPNModel`],[`hifigan`,`SpeechT5HifiGan`],[`efficientnet`,`EfficientNetModel`],[`decision_transformer`,`DecisionTransformerModel`],[`patchtst`,`PatchTSTModel`],[`patchtsmixer`,`PatchTSMixerModel`],[`mobilenet_v1`,`MobileNetV1Model`],[`mobilenet_v2`,`MobileNetV2Model`],[`mobilenet_v3`,`MobileNetV3Model`],[`mobilenet_v4`,`MobileNetV4Model`],[`maskformer`,`MaskFormerModel`],[`mgp-str`,`MgpstrForSceneTextRecognition`],[`style_text_to_speech_2`,`StyleTextToSpeech2Model`],[`openai_privacy_filter`,`OpenAIPrivacyFilterModel`]]),KD=new Map([[`t5`,`T5Model`],[`longt5`,`LongT5Model`],[`mt5`,`MT5Model`],[`bart`,`BartModel`],[`mbart`,`MBartModel`],[`marian`,`MarianModel`],[`whisper`,`WhisperModel`],[`cohere_asr`,`CohereAsrModel`],[`m2m_100`,`M2M100Model`],[`blenderbot`,`BlenderbotModel`],[`blenderbot-small`,`BlenderbotSmallModel`]]),qD=new Map([[`mimi`,`MimiModel`],[`dac`,`DacModel`],[`snac`,`SnacModel`]]),JD=new Map([[`bloom`,`BloomModel`],[`jais`,`JAISModel`],[`gpt2`,`GPT2Model`],[`gpt_oss`,`GptOssModel`],[`gptj`,`GPTJModel`],[`gpt_bigcode`,`GPTBigCodeModel`],[`gpt_neo`,`GPTNeoModel`],[`gpt_neox`,`GPTNeoXModel`],[`codegen`,`CodeGenModel`],[`llama`,`LlamaModel`],[`apertus`,`ApertusModel`],[`nanochat`,`NanoChatModel`],[`arcee`,`ArceeModel`],[`afmoe`,`AfmoeModel`],[`lfm2`,`Lfm2Model`],[`lfm2_moe`,`Lfm2MoeModel`],[`smollm3`,`SmolLM3Model`],[`exaone`,`ExaoneModel`],[`olmo`,`OlmoModel`],[`olmo2`,`Olmo2Model`],[`olmo3`,`Olmo3Model`],[`olmo_hybrid`,`OlmoHybridModel`],[`mobilellm`,`MobileLLMModel`],[`granite`,`GraniteModel`],[`granitemoehybrid`,`GraniteMoeHybridModel`],[`cohere`,`CohereModel`],[`cohere2`,`Cohere2Model`],[`gemma`,`GemmaModel`],[`gemma2`,`Gemma2Model`],[`vaultgemma`,`VaultGemmaModel`],[`gemma3_text`,`Gemma3Model`],[`helium`,`HeliumModel`],[`glm`,`GlmModel`],[`glm_moe_dsa`,`GlmMoeDsaModel`],[`openelm`,`OpenELMModel`],[`qwen2`,`Qwen2Model`],[`qwen2_moe`,`Qwen2MoeModel`],[`qwen3`,`Qwen3Model`],[`qwen3_moe`,`Qwen3MoeModel`],[`qwen3_next`,`Qwen3NextModel`],[`phi`,`PhiModel`],[`phi3`,`Phi3Model`],[`mpt`,`MptModel`],[`opt`,`OPTModel`],[`mistral`,`MistralModel`],[`mistral4`,`Mistral4Model`],[`ministral`,`MinistralModel`],[`ministral3`,`Ministral3Model`],[`ernie4_5`,`Ernie4_5ForCausalLM`],[`starcoder2`,`Starcoder2Model`],[`deepseek_v3`,`DeepseekV3Model`],[`falcon`,`FalconModel`],[`falcon_h1`,`FalconH1Model`],[`nemotron_h`,`NemotronHModel`],[`solar_open`,`SolarOpenModel`],[`stablelm`,`StableLmModel`],[`modernbert-decoder`,`ModernBertDecoderModel`],[`hunyuan_v1_dense`,`HunYuanDenseV1Model`],[`youtu`,`YoutuModel`]]),YD=new Map([[`speecht5`,`SpeechT5ForSpeechToText`],[`whisper`,`WhisperForConditionalGeneration`],[`lite-whisper`,`LiteWhisperForConditionalGeneration`],[`moonshine`,`MoonshineForConditionalGeneration`],[`cohere_asr`,`CohereAsrForConditionalGeneration`]]),XD=new Map([[`speecht5`,`SpeechT5ForTextToSpeech`]]),ZD=new Map([[`vits`,`VitsModel`],[`musicgen`,`MusicgenForConditionalGeneration`],[`supertonic`,`SupertonicForConditionalGeneration`]]),QD=new Map([[`bert`,`BertForSequenceClassification`],[`eurobert`,`EuroBertForSequenceClassification`],[`neobert`,`NeoBertForSequenceClassification`],[`modernbert`,`ModernBertForSequenceClassification`],[`roformer`,`RoFormerForSequenceClassification`],[`electra`,`ElectraForSequenceClassification`],[`esm`,`EsmForSequenceClassification`],[`convbert`,`ConvBertForSequenceClassification`],[`camembert`,`CamembertForSequenceClassification`],[`deberta`,`DebertaForSequenceClassification`],[`deberta-v2`,`DebertaV2ForSequenceClassification`],[`mpnet`,`MPNetForSequenceClassification`],[`albert`,`AlbertForSequenceClassification`],[`distilbert`,`DistilBertForSequenceClassification`],[`roberta`,`RobertaForSequenceClassification`],[`xlm`,`XLMForSequenceClassification`],[`xlm-roberta`,`XLMRobertaForSequenceClassification`],[`bart`,`BartForSequenceClassification`],[`mbart`,`MBartForSequenceClassification`],[`mobilebert`,`MobileBertForSequenceClassification`],[`squeezebert`,`SqueezeBertForSequenceClassification`]]),$D=new Map([[`bert`,`BertForTokenClassification`],[`eurobert`,`EuroBertForTokenClassification`],[`neobert`,`NeoBertForTokenClassification`],[`modernbert`,`ModernBertForTokenClassification`],[`roformer`,`RoFormerForTokenClassification`],[`electra`,`ElectraForTokenClassification`],[`esm`,`EsmForTokenClassification`],[`convbert`,`ConvBertForTokenClassification`],[`camembert`,`CamembertForTokenClassification`],[`deberta`,`DebertaForTokenClassification`],[`deberta-v2`,`DebertaV2ForTokenClassification`],[`mpnet`,`MPNetForTokenClassification`],[`distilbert`,`DistilBertForTokenClassification`],[`roberta`,`RobertaForTokenClassification`],[`xlm`,`XLMForTokenClassification`],[`xlm-roberta`,`XLMRobertaForTokenClassification`],[`openai_privacy_filter`,`OpenAIPrivacyFilterForTokenClassification`]]),eO=new Map([[`t5`,`T5ForConditionalGeneration`],[`longt5`,`LongT5ForConditionalGeneration`],[`mt5`,`MT5ForConditionalGeneration`],[`bart`,`BartForConditionalGeneration`],[`mbart`,`MBartForConditionalGeneration`],[`marian`,`MarianMTModel`],[`m2m_100`,`M2M100ForConditionalGeneration`],[`blenderbot`,`BlenderbotForConditionalGeneration`],[`blenderbot-small`,`BlenderbotSmallForConditionalGeneration`]]),tO=new Map([[`bloom`,`BloomForCausalLM`],[`gpt2`,`GPT2LMHeadModel`],[`gpt_oss`,`GptOssForCausalLM`],[`jais`,`JAISLMHeadModel`],[`gptj`,`GPTJForCausalLM`],[`gpt_bigcode`,`GPTBigCodeForCausalLM`],[`gpt_neo`,`GPTNeoForCausalLM`],[`gpt_neox`,`GPTNeoXForCausalLM`],[`codegen`,`CodeGenForCausalLM`],[`llama`,`LlamaForCausalLM`],[`nanochat`,`NanoChatForCausalLM`],[`apertus`,`ApertusForCausalLM`],[`llama4_text`,`Llama4ForCausalLM`],[`arcee`,`ArceeForCausalLM`],[`afmoe`,`AfmoeForCausalLM`],[`lfm2`,`Lfm2ForCausalLM`],[`lfm2_moe`,`Lfm2MoeForCausalLM`],[`smollm3`,`SmolLM3ForCausalLM`],[`exaone`,`ExaoneForCausalLM`],[`olmo`,`OlmoForCausalLM`],[`olmo2`,`Olmo2ForCausalLM`],[`olmo3`,`Olmo3ForCausalLM`],[`olmo_hybrid`,`OlmoHybridForCausalLM`],[`mobilellm`,`MobileLLMForCausalLM`],[`granite`,`GraniteForCausalLM`],[`granitemoehybrid`,`GraniteMoeHybridForCausalLM`],[`cohere`,`CohereForCausalLM`],[`cohere2`,`Cohere2ForCausalLM`],[`gemma`,`GemmaForCausalLM`],[`gemma2`,`Gemma2ForCausalLM`],[`vaultgemma`,`VaultGemmaForCausalLM`],[`gemma3_text`,`Gemma3ForCausalLM`],[`gemma3`,`Gemma3ForCausalLM`],[`helium`,`HeliumForCausalLM`],[`glm`,`GlmForCausalLM`],[`glm_moe_dsa`,`GlmMoeDsaForCausalLM`],[`openelm`,`OpenELMForCausalLM`],[`qwen2`,`Qwen2ForCausalLM`],[`qwen2_moe`,`Qwen2MoeForCausalLM`],[`qwen3`,`Qwen3ForCausalLM`],[`qwen3_moe`,`Qwen3MoeForCausalLM`],[`qwen3_next`,`Qwen3NextForCausalLM`],[`qwen2_vl`,`Qwen2VLForCausalLM`],[`qwen2_5_vl`,`Qwen2_5_VLForCausalLM`],[`qwen3_vl`,`Qwen3VLForCausalLM`],[`qwen3_vl_moe`,`Qwen3VLMoeForCausalLM`],[`qwen3_5`,`Qwen3_5ForCausalLM`],[`qwen3_5_text`,`Qwen3_5ForCausalLM`],[`qwen3_5_moe`,`Qwen3_5MoeForCausalLM`],[`gemma3n`,`Gemma3nForCausalLM`],[`gemma4`,`Gemma4ForCausalLM`],[`phi`,`PhiForCausalLM`],[`phi3`,`Phi3ForCausalLM`],[`mpt`,`MptForCausalLM`],[`opt`,`OPTForCausalLM`],[`mbart`,`MBartForCausalLM`],[`mistral`,`MistralForCausalLM`],[`mistral4`,`Mistral4ForCausalLM`],[`ministral`,`MinistralForCausalLM`],[`ministral3`,`Ministral3ForCausalLM`],[`ernie4_5`,`Ernie4_5ForCausalLM`],[`starcoder2`,`Starcoder2ForCausalLM`],[`deepseek_v3`,`DeepseekV3ForCausalLM`],[`falcon`,`FalconForCausalLM`],[`falcon_h1`,`FalconH1ForCausalLM`],[`nemotron_h`,`NemotronHForCausalLM`],[`trocr`,`TrOCRForCausalLM`],[`solar_open`,`SolarOpenForCausalLM`],[`stablelm`,`StableLmForCausalLM`],[`modernbert-decoder`,`ModernBertDecoderForCausalLM`],[`hunyuan_v1_dense`,`HunYuanDenseV1ForCausalLM`],[`youtu`,`YoutuForCausalLM`],[`phi3_v`,`Phi3VForCausalLM`]]),nO=new Map([[`multi_modality`,`MultiModalityCausalLM`]]),rO=new Map([[`bert`,`BertForMaskedLM`],[`eurobert`,`EuroBertForMaskedLM`],[`neobert`,`NeoBertForMaskedLM`],[`modernbert`,`ModernBertForMaskedLM`],[`roformer`,`RoFormerForMaskedLM`],[`electra`,`ElectraForMaskedLM`],[`esm`,`EsmForMaskedLM`],[`convbert`,`ConvBertForMaskedLM`],[`camembert`,`CamembertForMaskedLM`],[`deberta`,`DebertaForMaskedLM`],[`deberta-v2`,`DebertaV2ForMaskedLM`],[`mpnet`,`MPNetForMaskedLM`],[`albert`,`AlbertForMaskedLM`],[`distilbert`,`DistilBertForMaskedLM`],[`roberta`,`RobertaForMaskedLM`],[`xlm`,`XLMWithLMHeadModel`],[`xlm-roberta`,`XLMRobertaForMaskedLM`],[`mobilebert`,`MobileBertForMaskedLM`],[`squeezebert`,`SqueezeBertForMaskedLM`]]),iO=new Map([[`bert`,`BertForQuestionAnswering`],[`neobert`,`NeoBertForQuestionAnswering`],[`roformer`,`RoFormerForQuestionAnswering`],[`electra`,`ElectraForQuestionAnswering`],[`convbert`,`ConvBertForQuestionAnswering`],[`camembert`,`CamembertForQuestionAnswering`],[`deberta`,`DebertaForQuestionAnswering`],[`deberta-v2`,`DebertaV2ForQuestionAnswering`],[`mpnet`,`MPNetForQuestionAnswering`],[`albert`,`AlbertForQuestionAnswering`],[`distilbert`,`DistilBertForQuestionAnswering`],[`roberta`,`RobertaForQuestionAnswering`],[`xlm`,`XLMForQuestionAnswering`],[`xlm-roberta`,`XLMRobertaForQuestionAnswering`],[`mobilebert`,`MobileBertForQuestionAnswering`],[`squeezebert`,`SqueezeBertForQuestionAnswering`]]),aO=new Map([[`vision-encoder-decoder`,`VisionEncoderDecoderModel`],[`idefics3`,`Idefics3ForConditionalGeneration`],[`smolvlm`,`SmolVLMForConditionalGeneration`]]),oO=new Map([[`llava`,`LlavaForConditionalGeneration`],[`llava_onevision`,`LlavaOnevisionForConditionalGeneration`],[`moondream1`,`Moondream1ForConditionalGeneration`],[`florence2`,`Florence2ForConditionalGeneration`],[`qwen2_vl`,`Qwen2VLForConditionalGeneration`],[`qwen2_5_vl`,`Qwen2_5_VLForConditionalGeneration`],[`qwen3_vl`,`Qwen3VLForConditionalGeneration`],[`qwen3_vl_moe`,`Qwen3VLMoeForConditionalGeneration`],[`qwen3_5`,`Qwen3_5ForConditionalGeneration`],[`qwen3_5_moe`,`Qwen3_5MoeForConditionalGeneration`],[`lfm2_vl`,`Lfm2VlForConditionalGeneration`],[`idefics3`,`Idefics3ForConditionalGeneration`],[`smolvlm`,`SmolVLMForConditionalGeneration`],[`paligemma`,`PaliGemmaForConditionalGeneration`],[`llava_qwen2`,`LlavaQwen2ForCausalLM`],[`gemma3`,`Gemma3ForConditionalGeneration`],[`gemma3n`,`Gemma3nForConditionalGeneration`],[`gemma4`,`Gemma4ForConditionalGeneration`],[`mistral3`,`Mistral3ForConditionalGeneration`],[`lighton_ocr`,`LightOnOcrForConditionalGeneration`],[`glm_ocr`,`GlmOcrForConditionalGeneration`]]),sO=new Map([[`granite_speech`,`GraniteSpeechForConditionalGeneration`],[`ultravox`,`UltravoxModel`],[`voxtral`,`VoxtralForConditionalGeneration`],[`voxtral_realtime`,`VoxtralRealtimeForConditionalGeneration`]]),cO=new Map([[`vision-encoder-decoder`,`VisionEncoderDecoderModel`]]),lO=new Map([[`vit`,`ViTForImageClassification`],[`ijepa`,`IJepaForImageClassification`],[`pvt`,`PvtForImageClassification`],[`vit_msn`,`ViTMSNForImageClassification`],[`fastvit`,`FastViTForImageClassification`],[`mobilevit`,`MobileViTForImageClassification`],[`mobilevitv2`,`MobileViTV2ForImageClassification`],[`beit`,`BeitForImageClassification`],[`deit`,`DeiTForImageClassification`],[`hiera`,`HieraForImageClassification`],[`convnext`,`ConvNextForImageClassification`],[`convnextv2`,`ConvNextV2ForImageClassification`],[`dinov2`,`Dinov2ForImageClassification`],[`dinov2_with_registers`,`Dinov2WithRegistersForImageClassification`],[`resnet`,`ResNetForImageClassification`],[`swin`,`SwinForImageClassification`],[`segformer`,`SegformerForImageClassification`],[`efficientnet`,`EfficientNetForImageClassification`],[`mobilenet_v1`,`MobileNetV1ForImageClassification`],[`mobilenet_v2`,`MobileNetV2ForImageClassification`],[`mobilenet_v3`,`MobileNetV3ForImageClassification`],[`mobilenet_v4`,`MobileNetV4ForImageClassification`]]),uO=new Map([[`detr`,`DetrForObjectDetection`],[`rt_detr`,`RTDetrForObjectDetection`],[`rt_detr_v2`,`RTDetrV2ForObjectDetection`],[`rf_detr`,`RFDetrForObjectDetection`],[`d_fine`,`DFineForObjectDetection`],[`table-transformer`,`TableTransformerForObjectDetection`],[`yolos`,`YolosForObjectDetection`]]),dO=new Map([[`owlvit`,`OwlViTForObjectDetection`],[`owlv2`,`Owlv2ForObjectDetection`],[`grounding-dino`,`GroundingDinoForObjectDetection`]]),fO=new Map([[`detr`,`DetrForSegmentation`],[`clipseg`,`CLIPSegForImageSegmentation`]]),pO=new Map([[`segformer`,`SegformerForSemanticSegmentation`],[`sapiens`,`SapiensForSemanticSegmentation`],[`swin`,`SwinForSemanticSegmentation`],[`mobilenet_v1`,`MobileNetV1ForSemanticSegmentation`],[`mobilenet_v2`,`MobileNetV2ForSemanticSegmentation`],[`mobilenet_v3`,`MobileNetV3ForSemanticSegmentation`],[`mobilenet_v4`,`MobileNetV4ForSemanticSegmentation`]]),mO=new Map([[`detr`,`DetrForSegmentation`],[`maskformer`,`MaskFormerForInstanceSegmentation`]]),hO=new Map([[`sam`,`SamModel`],[`sam2`,`Sam2Model`],[`edgetam`,`EdgeTamModel`],[`sam3_tracker`,`Sam3TrackerModel`]]),gO=new Map([[`wav2vec2`,`Wav2Vec2ForCTC`],[`wav2vec2-bert`,`Wav2Vec2BertForCTC`],[`unispeech`,`UniSpeechForCTC`],[`unispeech-sat`,`UniSpeechSatForCTC`],[`wavlm`,`WavLMForCTC`],[`hubert`,`HubertForCTC`],[`parakeet_ctc`,`ParakeetForCTC`]]),_O=new Map([[`wav2vec2`,`Wav2Vec2ForSequenceClassification`],[`wav2vec2-bert`,`Wav2Vec2BertForSequenceClassification`],[`unispeech`,`UniSpeechForSequenceClassification`],[`unispeech-sat`,`UniSpeechSatForSequenceClassification`],[`wavlm`,`WavLMForSequenceClassification`],[`hubert`,`HubertForSequenceClassification`],[`audio-spectrogram-transformer`,`ASTForAudioClassification`]]),vO=new Map([[`wavlm`,`WavLMForXVector`]]),yO=new Map([[`unispeech-sat`,`UniSpeechSatForAudioFrameClassification`],[`wavlm`,`WavLMForAudioFrameClassification`],[`wav2vec2`,`Wav2Vec2ForAudioFrameClassification`],[`pyannote`,`PyAnnoteForAudioFrameClassification`]]),bO=new Map([[`vitmatte`,`VitMatteForImageMatting`]]),xO=new Map([[`patchtst`,`PatchTSTForPrediction`],[`patchtsmixer`,`PatchTSMixerForPrediction`]]),SO=new Map([[`swin2sr`,`Swin2SRForImageSuperResolution`]]),CO=new Map([[`chmv2`,`CHMv2ForDepthEstimation`],[`dpt`,`DPTForDepthEstimation`],[`depth_anything`,`DepthAnythingForDepthEstimation`],[`glpn`,`GLPNForDepthEstimation`],[`sapiens`,`SapiensForDepthEstimation`],[`depth_pro`,`DepthProForDepthEstimation`],[`metric3d`,`Metric3DForDepthEstimation`],[`metric3dv2`,`Metric3Dv2ForDepthEstimation`]]),wO=new Map([[`sapiens`,`SapiensForNormalEstimation`]]),TO=new Map([[`vitpose`,`VitPoseForPoseEstimation`]]),EO=new Map([[`clip`,`CLIPVisionModelWithProjection`],[`siglip`,`SiglipVisionModel`],[`jina_clip`,`JinaCLIPVisionModel`]]),DO=[[GD,X.EncoderOnly],[KD,X.EncoderDecoder],[JD,X.DecoderOnlyWithoutHead],[qD,X.AutoEncoder],[QD,X.EncoderOnly],[$D,X.EncoderOnly],[eO,X.Seq2Seq],[YD,X.Seq2Seq],[tO,X.DecoderOnly],[nO,X.MultiModality],[rO,X.EncoderOnly],[iO,X.EncoderOnly],[aO,X.Vision2Seq],[oO,X.ImageTextToText],[sO,X.AudioTextToText],[lO,X.EncoderOnly],[fO,X.EncoderOnly],[mO,X.EncoderOnly],[pO,X.EncoderOnly],[bO,X.EncoderOnly],[xO,X.EncoderOnly],[SO,X.EncoderOnly],[CO,X.EncoderOnly],[wO,X.EncoderOnly],[TO,X.EncoderOnly],[uO,X.EncoderOnly],[dO,X.EncoderOnly],[hO,X.MaskGeneration],[gO,X.EncoderOnly],[_O,X.EncoderOnly],[XD,X.Seq2Seq],[ZD,X.EncoderOnly],[vO,X.EncoderOnly],[yO,X.EncoderOnly],[EO,X.EncoderOnly]];for(let[e,t]of DO)for(let n of e.values()){wh.set(n,t);let e=qh[n];Eh.set(e,n),Th.set(n,e)}var OO=[[`MusicgenForConditionalGeneration`,sC,X.Musicgen],[`Phi3VForCausalLM`,dw,X.Phi3V],[`CLIPTextModelWithProjection`,e_,X.EncoderOnly],[`SiglipTextModel`,OT,X.EncoderOnly],[`JinaCLIPTextModel`,ox,X.EncoderOnly],[`ClapTextModelWithProjection`,Yg,X.EncoderOnly],[`ClapAudioModelWithProjection`,Xg,X.EncoderOnly],[`DacEncoderModel`,V_,X.EncoderOnly],[`DacDecoderModel`,H_,X.EncoderOnly],[`MimiEncoderModel`,Yx,X.EncoderOnly],[`MimiDecoderModel`,Xx,X.EncoderOnly],[`SnacEncoderModel`,IT,X.EncoderOnly],[`SnacDecoderModel`,LT,X.EncoderOnly],[`Gemma3nForConditionalGeneration`,Ly,X.ImageAudioTextToText],[`Gemma4ForConditionalGeneration`,zy,X.ImageAudioTextToText],[`SupertonicForConditionalGeneration`,oE,X.Supertonic],[`ChatterboxModel`,Hg,X.Chatterbox],[`VoxtralRealtimeForConditionalGeneration`,sD,X.VoxtralRealtime]];for(let[e,t,n]of OO)wh.set(e,n),Eh.set(t,e),Th.set(e,t);var kO=new Map([[`modnet`,fO],[`birefnet`,fO],[`isnet`,fO],[`ben`,fO]]);for(let[e,t]of kO.entries())t.set(e,`PreTrainedModel`),wh.set(e,X.EncoderOnly),Th.set(e,Z);var AO=new Set(kO.keys());wh.set(`PreTrainedModel`,X.EncoderOnly),Eh.set(Z,`PreTrainedModel`);var Q={MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING_NAMES:QD,MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING_NAMES:$D,MODEL_FOR_TEXT_TO_SPECTROGRAM_MAPPING_NAMES:XD,MODEL_FOR_TEXT_TO_WAVEFORM_MAPPING_NAMES:ZD,MODEL_FOR_MASKED_LM_MAPPING_NAMES:rO,MODEL_FOR_QUESTION_ANSWERING_MAPPING_NAMES:iO,MODEL_FOR_IMAGE_CLASSIFICATION_MAPPING_NAMES:lO,MODEL_FOR_IMAGE_SEGMENTATION_MAPPING_NAMES:fO,MODEL_FOR_SEMANTIC_SEGMENTATION_MAPPING_NAMES:pO,MODEL_FOR_UNIVERSAL_SEGMENTATION_MAPPING_NAMES:mO,MODEL_FOR_OBJECT_DETECTION_MAPPING_NAMES:uO,MODEL_FOR_ZERO_SHOT_OBJECT_DETECTION_MAPPING_NAMES:dO,MODEL_FOR_MASK_GENERATION_MAPPING_NAMES:hO,MODEL_FOR_CTC_MAPPING_NAMES:gO,MODEL_FOR_AUDIO_CLASSIFICATION_MAPPING_NAMES:_O,MODEL_FOR_AUDIO_XVECTOR_MAPPING_NAMES:vO,MODEL_FOR_AUDIO_FRAME_CLASSIFICATION_MAPPING_NAMES:yO,MODEL_FOR_DOCUMENT_QUESTION_ANSWERING_MAPPING_NAMES:cO,MODEL_FOR_IMAGE_MATTING_MAPPING_NAMES:bO,MODEL_FOR_IMAGE_TO_IMAGE_MAPPING_NAMES:SO,MODEL_FOR_DEPTH_ESTIMATION_MAPPING_NAMES:CO,MODEL_FOR_NORMAL_ESTIMATION_MAPPING_NAMES:wO,MODEL_FOR_POSE_ESTIMATION_MAPPING_NAMES:TO,MODEL_FOR_IMAGE_FEATURE_EXTRACTION_MAPPING_NAMES:EO,MODEL_FOR_IMAGE_TEXT_TO_TEXT_MAPPING_NAMES:oO,MODEL_FOR_AUDIO_TEXT_TO_TEXT_MAPPING_NAMES:sO,MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING_NAMES:eO,MODEL_FOR_SPEECH_SEQ_2_SEQ_MAPPING_NAMES:YD,MODEL_FOR_CAUSAL_LM_MAPPING_NAMES:tO,MODEL_FOR_VISION_2_SEQ_MAPPING_NAMES:aO};yh(Q);var $=class{static MODEL_CLASS_MAPPINGS=null;static BASE_IF_FAIL=!1;static supports(e){if(!this.MODEL_CLASS_MAPPINGS)return!1;for(let t of this.MODEL_CLASS_MAPPINGS)if(t.has(e))return!0;return this.BASE_IF_FAIL}static async from_pretrained(e,{progress_callback:t=null,config:n=null,cache_dir:r=null,local_files_only:i=!1,revision:a=`main`,model_file_name:o=null,subfolder:s=`onnx`,device:c=null,dtype:l=null,use_external_data_format:u=null,session_options:d={}}={}){let f={progress_callback:t,config:n,cache_dir:r,local_files_only:i,revision:a,model_file_name:o,subfolder:s,device:c,dtype:l,use_external_data_format:u,session_options:d};if(f.config=await Dm.from_pretrained(e,f),!this.MODEL_CLASS_MAPPINGS)throw Error("`MODEL_CLASS_MAPPINGS` not implemented for this type of `AutoClass`: "+this.name);let{model_type:p}=f.config;for(let t of this.MODEL_CLASS_MAPPINGS){let n=t.get(p);if(!n){for(let e of t.values())if(e[0]===p){n=e;break}if(!n)continue}return await qh[n].from_pretrained(e,f)}if(this.BASE_IF_FAIL)return AO.has(p)||N.warn(`Unknown model class "${p}", attempting to construct from base class.`),await Z.from_pretrained(e,f);throw Error(`Unsupported model type: ${p}`)}},jO=class extends ${static MODEL_CLASS_MAPPINGS=DO.map(e=>e[0]);static BASE_IF_FAIL=!0},MO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_SEQUENCE_CLASSIFICATION_MAPPING_NAMES]},NO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING_NAMES]},PO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING_NAMES]},FO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_SPEECH_SEQ_2_SEQ_MAPPING_NAMES]},IO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_TEXT_TO_SPECTROGRAM_MAPPING_NAMES]},LO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_TEXT_TO_WAVEFORM_MAPPING_NAMES]},RO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_CAUSAL_LM_MAPPING_NAMES]},zO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_MASKED_LM_MAPPING_NAMES]},BO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_QUESTION_ANSWERING_MAPPING_NAMES]},VO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_VISION_2_SEQ_MAPPING_NAMES]},HO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_CLASSIFICATION_MAPPING_NAMES]},UO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_SEGMENTATION_MAPPING_NAMES]},WO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_SEMANTIC_SEGMENTATION_MAPPING_NAMES]},GO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_UNIVERSAL_SEGMENTATION_MAPPING_NAMES]},KO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_OBJECT_DETECTION_MAPPING_NAMES]},qO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_ZERO_SHOT_OBJECT_DETECTION_MAPPING_NAMES]};(class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_MASK_GENERATION_MAPPING_NAMES]});var JO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_CTC_MAPPING_NAMES]},YO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_AUDIO_CLASSIFICATION_MAPPING_NAMES]};(class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_AUDIO_XVECTOR_MAPPING_NAMES]}),class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_AUDIO_FRAME_CLASSIFICATION_MAPPING_NAMES]};var XO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_DOCUMENT_QUESTION_ANSWERING_MAPPING_NAMES]};(class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_MATTING_MAPPING_NAMES]});var ZO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_TO_IMAGE_MAPPING_NAMES]},QO=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_DEPTH_ESTIMATION_MAPPING_NAMES]};(class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_NORMAL_ESTIMATION_MAPPING_NAMES]}),class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_POSE_ESTIMATION_MAPPING_NAMES]};var $O=class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_FEATURE_EXTRACTION_MAPPING_NAMES]};(class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_IMAGE_TEXT_TO_TEXT_MAPPING_NAMES]}),class extends ${static MODEL_CLASS_MAPPINGS=[Q.MODEL_FOR_AUDIO_TEXT_TO_TEXT_MAPPING_NAMES]};async function ek(e){return Array.isArray(e)||(e=[e]),await Promise.all(e.map(e=>Ud.read(e)))}async function tk(e,t){return Array.isArray(e)||(e=[e]),await Promise.all(e.map(e=>typeof e==`string`||e instanceof URL?qu(e,t):e instanceof Float64Array?new Float32Array(e):e))}function nk(e,t){t&&(e=e.map(e=>e|0));let[n,r,i,a]=e;return{xmin:n,ymin:r,xmax:i,ymax:a}}var rk=class extends ni{constructor({task:e,model:t,tokenizer:n=null,processor:r=null}){super(),this.task=e,this.model=t,this.tokenizer=n,this.processor=r}async dispose(){await this.model.dispose()}},ik=class extends rk{async _call(e,{top_k:t=1}={}){let n=this.tokenizer(e,{padding:!0,truncation:!0}),r=await this.model(n),{problem_type:i,id2label:a}=this.model.config,o=i===`multi_label_classification`?e=>e.sigmoid():e=>new U(`float32`,oc(e.data),e.dims),s=[];for(let e of r.logits){let n=await al(o(e),t),r=n[0].tolist(),i=n[1].tolist().map((e,t)=>({label:a?a[e]:`LABEL_${e}`,score:r[t]}));t===1?s.push(...i):s.push(i)}return Array.isArray(e)||t===1?s:s[0]}},ak=class extends rk{async _call(e,{ignore_labels:t=[`O`],aggregation_strategy:n=`none`}={}){if(n!==`none`&&n!==`simple`)throw Error(`Invalid aggregation_strategy: "${n}". Must be one of "none" or "simple".`);let r=Array.isArray(e),i=this.tokenizer(r?e:[e],{padding:!0,truncation:!0}),a=(await this.model(i)).logits,o=this.model.config.id2label,s=[];for(let e=0;e<a.dims[0];++e){let r=i.input_ids[e].tolist(),c=a[e],l=[];for(let e=0;e<c.dims[0];++e){let n=c[e],i=lc(n.data)[1],a=o?o[i]:`LABEL_${i}`;if(t.includes(a))continue;let s=this.tokenizer.decode([r[e]],{skip_special_tokens:!0});if(s===``)continue;let u=oc(n.data);l.push({entity:a,score:u[i],index:e,word:s})}s.push(n===`simple`?sk(l,r,this.tokenizer):l)}return r?s:s[0]}};function ok(e){let t=e[0];return e[1]===`-`&&(t===`B`||t===`I`||t===`E`||t===`S`)?[t,e.slice(2)]:[`I`,e]}function sk(e,t,n){let r=[],i=null;for(let t=0;t<e.length;++t){let[n,a]=ok(e[t].entity);i===a&&n!==`B`&&n!==`S`?(r[r.length-1].end=t+1,n===`E`&&(i=null)):(r.push({tag:a,start:t,end:t+1}),i=n===`S`?null:a)}return r.map(({tag:r,start:i,end:a})=>{let o=0,s=[];for(let n=i;n<a;++n)o+=e[n].score,s.push(t[e[n].index]);return{entity_group:r,score:o/(a-i),word:n.decode(s,{skip_special_tokens:!0})}})}var ck=class extends rk{async _call(e,t,{top_k:n=1}={}){let r=this.tokenizer(e,{text_pair:t,padding:!0,truncation:!0}),i=Array.isArray(e),{start_logits:a,end_logits:o}=await this.model(r),s=r.input_ids.tolist(),c=r.attention_mask.tolist(),{all_special_ids:l,sep_token_id:u}=this.tokenizer,d=[];for(let e=0;e<a.dims[0];++e){let t=s[e],r=t.findIndex(e=>e==u),i=a[e].tolist(),f=o[e].tolist();for(let n=1;n<i.length;++n)(c[e]==0||n<=r||l.findIndex(e=>e==t[n])!==-1)&&(i[n]=-1/0,f[n]=-1/0);let p=oc(i).map((e,t)=>[e,t]),m=oc(f).map((e,t)=>[e,t]);p[0][0]=0,m[0][0]=0;let h=li(p,m).filter(e=>e[0][1]<=e[1][1]).map(e=>[e[0][1],e[1][1],e[0][0]*e[1][0]]).sort((e,t)=>t[2]-e[2]),g=[];for(let e=0;e<Math.min(h.length,n);++e){let[n,r,i]=h[e],a=t.slice(n,r+1),o=this.tokenizer.decode(a,{skip_special_tokens:!0});g.push({answer:o,score:i})}n===1?d.push(...g):d.push(g)}return i?d:d[0]}},lk=class extends rk{async _call(e,{top_k:t=5}={}){let{mask_token_id:n,mask_token:r}=this.tokenizer,i=this.tokenizer(e,{padding:!0,truncation:!0}),{logits:a}=await this.model(i),o=[],s=i.input_ids.tolist();for(let e=0;e<s.length;++e){let i=s[e],c=i.findIndex(e=>e==n);if(c===-1)throw Error(`Mask token (${r}) not found in text.`);let l=a[e][c],u=await al(new U(`float32`,oc(l.data),l.dims),t),d=u[0].tolist(),f=u[1].tolist();o.push(f.map((e,t)=>{let n=i.slice();return n[c]=e,{score:d[t],token:Number(e),token_str:this.tokenizer.decode([e]),sequence:this.tokenizer.decode(n,{skip_special_tokens:!0})}}))}return Array.isArray(e)?o:o[0]}},uk=class extends rk{_default_generation_config={max_new_tokens:256};_key=`generated_text`;async _call(e,t={}){Array.isArray(e)||(e=[e]),this.model.config.prefix&&(e=e.map(e=>this.model.config.prefix+e));let n=this.model.config.task_specific_params;n&&n[this.task]&&n[this.task].prefix&&(e=e.map(e=>n[this.task].prefix+e));let r=this.tokenizer,i={padding:!0,truncation:!0},a;a=this.task===`translation`&&`_build_translation_inputs`in r?r._build_translation_inputs(e,i,t):r(e,i);let o=await this.model.generate({...a,...this._default_generation_config,...t});return r.batch_decode(o,{skip_special_tokens:!0}).map(e=>({[this._key]:e}))}},dk=class extends uk{_key=`summary_text`},fk=class extends uk{_key=`translation_text`};function pk(e){return Array.isArray(e)&&e.every(e=>`role`in e&&`content`in e)}var mk=class extends rk{_default_generation_config={max_new_tokens:256};async _call(e,t={}){let{add_special_tokens:n,return_full_text:r,tools:i,documents:a,chat_template:o,tokenizer_encode_kwargs:s,...c}=t,l=!1,u=!1,d=n??(this.tokenizer.add_bos_token||this.tokenizer.add_eos_token)??!1,f=s,p;if(typeof e==`string`)p=e=[e];else if(Array.isArray(e)&&e.every(e=>typeof e==`string`))l=!0,p=e;else{if(pk(e))e=[e];else if(Array.isArray(e)&&e.every(pk))l=!0;else throw Error(`Input must be a string, an array of strings, a Chat, or an array of Chats`);u=!0;let t={tokenize:!1,add_generation_prompt:!0,...di({tools:i,documents:a,chat_template:o},[`tools`,`documents`,`chat_template`]),...f};p=e.map(e=>this.tokenizer.apply_chat_template(e,t)),d=!1,f=void 0}let m=u?!1:r??!0;this.tokenizer.padding_side=`left`;let h=this.tokenizer(p,{add_special_tokens:d,padding:!0,truncation:!0,...f}),g=await this.model.generate({...h,...this._default_generation_config,...c}),_=this.tokenizer.batch_decode(g,{skip_special_tokens:!0}),v;!m&&h.input_ids.dims.at(-1)>0&&(v=this.tokenizer.batch_decode(h.input_ids,{skip_special_tokens:!0}).map(e=>e.length));let y=Array.from({length:e.length},e=>[]);for(let t=0;t<_.length;++t){let n=Math.floor(t/g.dims[0]*e.length);v&&(_[t]=_[t].slice(v[n])),y[n].push({generated_text:u?[...e[n],{role:`assistant`,content:_[t]}]:_[t]})}return!l&&y.length===1?y[0]:y}},hk=class extends rk{constructor(e){super(e),this.label2id=Object.fromEntries(Object.entries(this.model.config.label2id).map(([e,t])=>[e.toLowerCase(),t])),this.entailment_id=this.label2id.entailment,this.entailment_id===void 0&&(N.warn(`Could not find 'entailment' in label2id mapping. Using 2 as entailment_id.`),this.entailment_id=2),this.contradiction_id=this.label2id.contradiction??this.label2id.not_entailment,this.contradiction_id===void 0&&(N.warn(`Could not find 'contradiction' in label2id mapping. Using 0 as contradiction_id.`),this.contradiction_id=0)}async _call(e,t,{hypothesis_template:n=`This example is {}.`,multi_label:r=!1}={}){let i=Array.isArray(e);i||(e=[e]),Array.isArray(t)||(t=[t]);let a=t.map(e=>n.replace(`{}`,e)),o=r||t.length===1,s=[];for(let n of e){let e=[];for(let t of a){let r=this.tokenizer(n,{text_pair:t,padding:!0,truncation:!0}),i=await this.model(r);o?e.push([i.logits.data[this.contradiction_id],i.logits.data[this.entailment_id]]):e.push(i.logits.data[this.entailment_id])}let r=(o?e.map(e=>oc(e)[1]):oc(e)).map((e,t)=>[e,t]).sort((e,t)=>t[0]-e[0]);s.push({sequence:n,labels:r.map(e=>t[e[1]]),scores:r.map(e=>e[0])})}return i?s:s[0]}},gk=class extends rk{async _call(e,{top_k:t=5}={}){let n=this.processor.feature_extractor.config.sampling_rate,r=await tk(e,n),i=this.model.config.id2label,a=[];for(let e of r){let n=await this.processor(e),r=(await this.model(n)).logits[0],o=await al(new U(`float32`,oc(r.data),r.dims),t),s=o[0].tolist(),c=o[1].tolist().map((e,t)=>({label:i?i[e]:`LABEL_${e}`,score:s[t]}));a.push(c)}return Array.isArray(e)?a:a[0]}},_k=class extends rk{async _call(e,t,{hypothesis_template:n=`This is a sound of {}.`}={}){let r=!Array.isArray(e);r&&(e=[e]);let i=t.map(e=>n.replace(`{}`,e)),a=this.tokenizer(i,{padding:!0,truncation:!0}),o=this.processor.feature_extractor.config.sampling_rate,s=await tk(e,o),c=[];for(let e of s){let n=await this.processor(e),r=oc((await this.model({...a,...n})).logits_per_audio.data);c.push([...r].map((e,n)=>({score:e,label:t[n]})))}return r?c[0]:c}},vk=class extends rk{_default_generation_config={};async _call(e,t={}){switch(t={...this._default_generation_config,...t},this.model.config.model_type){case`whisper`:case`lite-whisper`:return this._call_whisper(e,t);case`wav2vec2`:case`wav2vec2-bert`:case`unispeech`:case`unispeech-sat`:case`hubert`:case`parakeet_ctc`:return this._call_wav2vec2(e,t);case`moonshine`:return this._call_moonshine(e,t);case`cohere_asr`:return this._call_cohere_asr(e,t);default:throw Error(`AutomaticSpeechRecognitionPipeline does not support model type '${this.model.config.model_type}'.`)}}async _call_wav2vec2(e,t){t.language&&N.warn('`language` parameter is not yet supported for `wav2vec2` models, defaulting to "English".'),t.task&&N.warn('`task` parameter is not yet supported for `wav2vec2` models, defaulting to "transcribe".');let n=!Array.isArray(e),r=n?[e]:e,i=this.processor.feature_extractor.config.sampling_rate,a=await tk(r,i),o=[];for(let e of a){let t=await this.processor(e),n=(await this.model(t)).logits[0],r=[];for(let e of n)r.push(lc(e.data)[1]);let i=this.tokenizer.decode(r,{skip_special_tokens:!0}).trim();o.push({text:i})}return n?o[0]:o}async _call_whisper(e,t){let n=t.return_timestamps??!1,r=t.chunk_length_s??0,i=t.force_full_sequences??!1,a=t.stride_length_s??null,o={...t};n===`word`&&(o.return_token_timestamps=!0,o.return_timestamps=!0);let s=!Array.isArray(e),c=s?[e]:e,l=this.processor.feature_extractor.config,u=l.chunk_length/this.model.config.max_source_positions,d=l.hop_length,f=l.sampling_rate,p=await tk(c,f),m=[];for(let e of p){let t=[];if(r>0){if(a===null)a=r/6;else if(r<=a)throw Error("`chunk_length_s` must be larger than `stride_length_s`.");let n=f*r,i=f*a,o=n-2*i,s=0;for(;;){let r=s+n,a=e.subarray(s,r),c=await this.processor(a),l=s===0,u=r>=e.length;if(t.push({stride:[a.length,l?0:i,u?0:i],input_features:c.input_features,is_last:u}),u)break;s+=o}}else t=[{stride:[e.length,0,0],input_features:(await this.processor(e)).input_features,is_last:!0}];for(let e of t){o.num_frames=Math.floor(e.stride[0]/d);let t=await this.model.generate({inputs:e.input_features,...o});if(n===`word`){let n=t.sequences.tolist()[0],r=t.token_timestamps.tolist()[0],i=this.tokenizer.timestamp_begin,a=Math.max(n.findIndex(e=>Number(e)>=i),0);e.tokens=n.slice(a),e.token_timestamps=r.slice(a).map(e=>hc(e,2))}else e.tokens=t[0].tolist();e.stride=e.stride.map(e=>e/f)}let[s,c]=this.tokenizer._decode_asr(t,{time_precision:u,return_timestamps:n,force_full_sequences:i});m.push({text:s,...c})}return s?m[0]:m}async _call_moonshine(e,t){let n=!Array.isArray(e),r=n?[e]:e,i=this.processor.feature_extractor.config.sampling_rate,a=await tk(r,i),o=[];for(let e of a){let n=await this.processor(e),r=Math.floor(e.length/i)*6,a=await this.model.generate({max_new_tokens:r,...t,...n}),s=this.processor.batch_decode(a,{skip_special_tokens:!0})[0];o.push({text:s})}return n?o[0]:o}async _call_cohere_asr(e,t){let n=!Array.isArray(e),r=n?[e]:e,i=this.processor.feature_extractor,a=i.config.sampling_rate,o=await tk(r,a),s=t.language??`en`,c=this.processor.get_decoder_prompt_ids(s),l=[];for(let e of o){let n=i.split_audio(e),r=[];for(let e of n){let n=await this.processor(e),i=await this.model.generate({...n,decoder_input_ids:c,...t}),a=this.tokenizer.decode(i[0].tolist(),{skip_special_tokens:!0}).trim();r.push(a)}let a=this.processor.constructor.join_chunks(r,s);l.push({text:a})}return n?l[0]:l}},yk=class extends rk{DEFAULT_VOCODER_ID=`Xenova/speecht5_hifigan`;constructor(e){super(e),this.vocoder=e.vocoder??null}async _prepare_speaker_embeddings(e,t){if((typeof e==`string`||e instanceof URL)&&(e=new Float32Array(await(await M.fetch(e)).arrayBuffer())),e instanceof Float32Array)e=new U(`float32`,e,[e.length]);else if(!(e instanceof U))throw Error("Speaker embeddings must be a `Tensor`, `Float32Array`, `string`, or `URL`.");if(t>1){if(e.dims[0]===1)e=e.repeat(t,1);else if(e.dims[0]!==t)throw Error(`Expected speaker embeddings batch size to be 1 or ${t}, but got ${e.dims[0]}.`)}return e}_postprocess_waveform(e,t,n,r=null){let i=t.data,[a,o]=t.dims,s=r?r.data:null,c=[];for(let e=0;e<a;++e){let t=s?Math.min(Math.ceil(s[e]),o):o,r=e*o;c.push(new fd(i.slice(r,r+t),n))}return Array.isArray(e)?c:c[0]}async _call(e,t){return this.processor?this._call_text_to_spectrogram(e,t):this.model.config.model_type===`supertonic`?this._call_supertonic(e,t):this._call_text_to_waveform(e)}async _call_supertonic(e,{speaker_embeddings:t,num_inference_steps:n,speed:r}){if(!t)throw Error(`Speaker embeddings must be provided for Supertonic models.`);let{sampling_rate:i,style_dim:a}=this.model.config,o=this.tokenizer(e,{padding:!0,truncation:!0}),s=o.input_ids.dims[0];t=await this._prepare_speaker_embeddings(t,s),t=t.view(s,-1,a);let{waveform:c,durations:l}=await this.model.generate_speech({...o,style:t,num_inference_steps:n,speed:r});return this._postprocess_waveform(e,c,i,l)}async _call_text_to_waveform(e){let t=this.tokenizer(e,{padding:!0,truncation:!0}),{waveform:n}=await this.model(t),r=this.model.config.sampling_rate;return this._postprocess_waveform(e,n,r)}async _call_text_to_spectrogram(e,{speaker_embeddings:t}){this.vocoder||=(N.info(`No vocoder specified, using default HifiGan vocoder.`),await jO.from_pretrained(this.DEFAULT_VOCODER_ID,{dtype:`fp32`}));let{input_ids:n}=this.tokenizer(e,{padding:!0,truncation:!0}),r=n.dims[0];t=await this._prepare_speaker_embeddings(t,r),t=t.view(r,-1);let{waveform:i}=await this.model.generate_speech(n,t,{vocoder:this.vocoder}),a=this.processor.feature_extractor.config.sampling_rate;return this._postprocess_waveform(e,i,a)}},bk=class extends rk{async _call(e,t={}){let n=Array.isArray(e),r=await ek(e),{pixel_values:i}=await this.processor(r),a=[];for(let e of i){e.dims=[1,...e.dims];let n=await this.model.generate({inputs:e,...t}),r=this.tokenizer.batch_decode(n,{skip_special_tokens:!0}).map(e=>({generated_text:e.trim()}));a.push(r)}return n?a:a[0]}},xk=class extends rk{async _call(e,{top_k:t=5}={}){let n=await ek(e),{pixel_values:r}=await this.processor(n),i=await this.model({pixel_values:r}),{id2label:a}=this.model.config,o=[];for(let e of i.logits){let n=await al(new U(`float32`,oc(e.data),e.dims),t),r=n[0].tolist(),i=n[1].tolist().map((e,t)=>({label:a?a[e]:`LABEL_${e}`,score:r[t]}));o.push(i)}return Array.isArray(e)?o:o[0]}},Sk={panoptic:`post_process_panoptic_segmentation`,instance:`post_process_instance_segmentation`,semantic:`post_process_semantic_segmentation`},Ck=class extends rk{async _call(e,{threshold:t=.5,mask_threshold:n=.5,overlap_mask_area_threshold:r=.8,label_ids_to_fuse:i=null,target_sizes:a=null,subtask:o=null}={}){if(Array.isArray(e)&&e.length!==1)throw Error(`Image segmentation pipeline currently only supports a batch size of 1.`);let s=await ek(e),c=s.map(e=>[e.height,e.width]),l=await this.processor(s),{inputNames:u,outputNames:d}=this.model.sessions.model;if(!u.includes(`pixel_values`)){if(u.length!==1)throw Error(`Expected a single input name, but got ${u.length} inputs: ${u}.`);let e=u[0];if(e in l)throw Error(`Input name ${e} already exists in the inputs.`);l[e]=l.pixel_values}let f=await this.model(l),p=null;if(o!==null)p=Sk[o];else if(this.processor.image_processor){for(let[e,t]of Object.entries(Sk))if(t in this.processor.image_processor){p=this.processor.image_processor[t].bind(this.processor.image_processor),o=e;break}}let m=this.model.config.id2label,h=[];if(!o){let e=f[d[0]];for(let t=0;t<c.length;++t){let n=c[t],r=e[t];r.data.some(e=>e<-1e-5||e>1.00001)&&r.sigmoid_();let i=await Ud.fromTensor(r.mul_(255).to(`uint8`)).resize(n[1],n[0]);h.push({label:null,score:null,mask:i})}}else if(o===`panoptic`||o===`instance`){let e=p(f,t,n,r,i,a??c)[0],o=e.segmentation;for(let t of e.segments_info){let e=new Uint8ClampedArray(o.data.length);for(let n=0;n<o.data.length;++n)o.data[n]===t.id&&(e[n]=255);let n=new Ud(e,o.dims[1],o.dims[0],1);h.push({score:t.score,label:m[t.label_id],mask:n})}}else if(o===`semantic`){let{segmentation:e,labels:t}=p(f,a??c)[0];for(let n of t){let t=new Uint8ClampedArray(e.data.length);for(let r=0;r<e.data.length;++r)e.data[r]===n&&(t[r]=255);let r=new Ud(t,e.dims[1],e.dims[0],1);h.push({score:null,label:m[n],mask:r})}}else throw Error(`Subtask ${o} not supported.`);return h}};Object.freeze({"text-classification":{pipeline:ik,model:MO,default:{model:`Xenova/distilbert-base-uncased-finetuned-sst-2-english`},type:`text`},"token-classification":{pipeline:ak,model:NO,default:{model:`Xenova/bert-base-multilingual-cased-ner-hrl`},type:`text`},"question-answering":{pipeline:ck,model:BO,default:{model:`Xenova/distilbert-base-cased-distilled-squad`},type:`text`},"fill-mask":{pipeline:lk,model:zO,default:{model:`onnx-community/ettin-encoder-32m-ONNX`,dtype:`fp32`},type:`text`},summarization:{pipeline:dk,model:PO,default:{model:`Xenova/distilbart-cnn-6-6`},type:`text`},translation:{pipeline:fk,model:PO,default:{model:`Xenova/t5-small`},type:`text`},"text2text-generation":{pipeline:uk,model:PO,default:{model:`Xenova/flan-t5-small`},type:`text`},"text-generation":{pipeline:mk,model:RO,default:{model:`onnx-community/Qwen3-0.6B-ONNX`,dtype:`q4`},type:`text`},"zero-shot-classification":{pipeline:hk,model:MO,default:{model:`Xenova/distilbert-base-uncased-mnli`},type:`text`},"audio-classification":{pipeline:gk,model:YO,default:{model:`Xenova/wav2vec2-base-superb-ks`},type:`audio`},"zero-shot-audio-classification":{pipeline:_k,model:jO,default:{model:`Xenova/clap-htsat-unfused`},type:`multimodal`},"automatic-speech-recognition":{pipeline:vk,model:[FO,JO],default:{model:`Xenova/whisper-tiny.en`},type:`multimodal`},"text-to-audio":{pipeline:yk,model:[LO,IO],default:{model:`onnx-community/Supertonic-TTS-ONNX`,dtype:`fp32`},type:`text`},"image-to-text":{pipeline:bk,model:VO,default:{model:`Xenova/vit-gpt2-image-captioning`},type:`multimodal`},"image-classification":{pipeline:xk,model:HO,default:{model:`Xenova/vit-base-patch16-224`},type:`multimodal`},"image-segmentation":{pipeline:Ck,model:[UO,WO,GO],default:{model:`Xenova/detr-resnet-50-panoptic`},type:`multimodal`},"background-removal":{pipeline:class extends Ck{async _call(e,t={}){let n=await ek(e),r=await super._call(e,t),i=n.map((e,t)=>{let n=e.clone();return n.putAlpha(r[t].mask),n});return Array.isArray(e)?i:i[0]}},model:[UO,WO,GO],default:{model:`Xenova/modnet`},type:`image`},"zero-shot-image-classification":{pipeline:class extends rk{async _call(e,t,{hypothesis_template:n=`This is a photo of {}`}={}){let r=Array.isArray(e),i=await ek(e),a=t.map(e=>n.replace(`{}`,e)),o=this.tokenizer(a,{padding:this.model.config.model_type===`siglip`?`max_length`:!0,truncation:!0}),{pixel_values:s}=await this.processor(i),c=await this.model({...o,pixel_values:s}),l=this.model.config.model_type===`siglip`?e=>e.sigmoid().data:e=>oc(e.data),u=[];for(let e of c.logits_per_image){let n=[...l(e)].map((e,n)=>({score:e,label:t[n]}));n.sort((e,t)=>t.score-e.score),u.push(n)}return r?u:u[0]}},model:jO,default:{model:`Xenova/clip-vit-base-patch32`},type:`multimodal`},"object-detection":{pipeline:class extends rk{async _call(e,{threshold:t=.9,percentage:n=!1}={}){let r=Array.isArray(e);if(r&&e.length!==1)throw Error(`Object detection pipeline currently only supports a batch size of 1.`);let i=await ek(e),a=n?null:i.map(e=>[e.height,e.width]),{pixel_values:o,pixel_mask:s}=await this.processor(i),c=await this.model({pixel_values:o,pixel_mask:s}),l=this.processor.image_processor.post_process_object_detection(c,t,a),{id2label:u}=this.model.config,d=l.map(e=>e.boxes.map((t,r)=>({score:e.scores[r],label:u[e.classes[r]],box:nk(t,!n)})));return r?d:d[0]}},model:KO,default:{model:`Xenova/detr-resnet-50`},type:`multimodal`},"zero-shot-object-detection":{pipeline:class extends rk{async _call(e,t,{threshold:n=.1,top_k:r=null,percentage:i=!1}={}){let a=Array.isArray(e),o=await ek(e),s=this.tokenizer(t,{padding:!0,truncation:!0}),c=await this.processor(o),l=[];for(let e=0;e<o.length;++e){let a=o[e],u=i?null:[[a.height,a.width]],d=c.pixel_values[e].unsqueeze_(0),f=await this.model({...s,pixel_values:d}),p;if(`post_process_grounded_object_detection`in this.processor){let e=this.processor.post_process_grounded_object_detection(f,s.input_ids,{box_threshold:n,text_threshold:n,target_sizes:u})[0];p=e.boxes.map((t,n)=>({score:e.scores[n],label:e.labels[n],box:nk(t,!i)}))}else{let e=this.processor.image_processor.post_process_object_detection(f,n,u,!0)[0];p=e.boxes.map((n,r)=>({score:e.scores[r],label:t[e.classes[r]],box:nk(n,!i)}))}p.sort((e,t)=>t.score-e.score),r!==null&&(p=p.slice(0,r)),l.push(p)}return a?l:l[0]}},model:qO,default:{model:`Xenova/owlvit-base-patch32`},type:`multimodal`},"document-question-answering":{pipeline:class extends rk{_default_generation_config={max_new_tokens:256};async _call(e,t,n={}){if(Array.isArray(e)){if(e.length!==1)throw Error(`Document Question Answering pipeline currently only supports a batch size of 1.`);e=e[0]}let r=(await ek(e))[0],{pixel_values:i}=await this.processor(r),a=`<s_docvqa><s_question>${t}</s_question><s_answer>`,o=this.tokenizer(a,{add_special_tokens:!1,padding:!0,truncation:!0}).input_ids,s=await this.model.generate({inputs:i,max_length:this.model.config.decoder.max_position_embeddings,decoder_input_ids:o,...this._default_generation_config,...n}),c=this.tokenizer.batch_decode(s)[0].match(/<s_answer>(.*?)<\/s_answer>/),l=null;return c&&c.length>=2&&(l=c[1].trim()),[{answer:l}]}},model:XO,default:{model:`Xenova/donut-base-finetuned-docvqa`},type:`multimodal`},"image-to-image":{pipeline:class extends rk{async _call(e){let t=await ek(e),n=await this.processor(t),r=await this.model(n),i=[];for(let e of r.reconstruction){let t=e.squeeze().clamp_(0,1).mul_(255).round_().to(`uint8`);i.push(Ud.fromTensor(t))}return Array.isArray(e)?i:i[0]}},model:ZO,default:{model:`Xenova/swin2SR-classical-sr-x2-64`},type:`image`},"depth-estimation":{pipeline:class extends rk{async _call(e){let t=await ek(e),n=await this.processor(t),{predicted_depth:r}=await this.model(n),i=[];for(let e=0;e<t.length;++e){let n=r[e],[a,o]=n.dims.slice(-2),[s,c]=t[e].size,l=(await rl(n.view(1,1,a,o),{size:[c,s],mode:`bilinear`})).view(c,s),u=l.min().item(),d=l.max().item(),f=l.sub(u).div_(d-u).mul_(255).to(`uint8`).unsqueeze(0),p=Ud.fromTensor(f);i.push({predicted_depth:l,depth:p})}return Array.isArray(e)?i:i[0]}},model:QO,default:{model:`onnx-community/depth-anything-v2-small`},type:`image`},"feature-extraction":{pipeline:class extends rk{async _call(e,{pooling:t=`none`,normalize:n=!1,quantize:r=!1,precision:i=`binary`}={}){let a=this.tokenizer(e,{padding:!0,truncation:!0}),o=await this.model(a),s=o.last_hidden_state??o.logits??o.token_embeddings;switch(t){case`none`:break;case`mean`:s=cl(s,a.attention_mask);break;case`first_token`:case`cls`:s=s.slice(null,0);break;case`last_token`:case`eos`:s=s.slice(null,-1);break;default:throw Error(`Pooling method '${t}' not supported.`)}return n&&(s=s.normalize(2,-1)),r&&(s=El(s,i)),s}},model:jO,default:{model:`onnx-community/all-MiniLM-L6-v2-ONNX`,dtype:`fp32`},type:`text`},"image-feature-extraction":{pipeline:class extends rk{async _call(e,{pool:t=null}={}){let n=await ek(e),{pixel_values:r}=await this.processor(n),i=await this.model({pixel_values:r}),a;if(t){if(!(`pooler_output`in i))throw Error(`No pooled output was returned. Make sure the model has a 'pooler' layer when using the 'pool' option.`);a=i.pooler_output}else a=i.last_hidden_state??i.logits??i.image_embeds;return a}},model:[$O,jO],default:{model:`onnx-community/dinov3-vits16-pretrain-lvd1689m-ONNX`,dtype:`fp32`},type:`image`}}),Object.freeze({"sentiment-analysis":`text-classification`,ner:`token-classification`,asr:`automatic-speech-recognition`,"text-to-speech":`text-to-audio`,embeddings:`feature-extraction`});var wk=e=>e>=19968&&e<=40959||e>=13312&&e<=19903||e>=131072&&e<=173791||e>=173824&&e<=177983||e>=177984&&e<=178207||e>=178208&&e<=183983||e>=63744&&e<=64255||e>=194560&&e<=195103,Tk=class{put(e){throw Error(`Not implemented`)}end(){throw Error(`Not implemented`)}},Ek=j.IS_PROCESS_AVAILABLE?e=>process.stdout.write(e):e=>console.log(e),Dk=class extends Tk{constructor(e,{skip_prompt:t=!1,callback_function:n=null,token_callback_function:r=null,skip_special_tokens:i=!0,decode_kwargs:a={},...o}={}){super(),this.tokenizer=e,this.skip_prompt=t,this.callback_function=n??Ek,this.token_callback_function=r,this.decode_kwargs={skip_special_tokens:i,...a,...o},this.token_cache=[],this.print_len=0,this.next_tokens_are_prompt=!0,this.special_ids=new Set(this.tokenizer.all_special_ids.map(BigInt))}put(e){if(e.length>1)throw Error(`TextStreamer only supports batch size of 1`);let t=this.next_tokens_are_prompt;if(t&&(this.next_tokens_are_prompt=!1,this.skip_prompt))return;let n=e[0];if(this.token_callback_function?.(n),n.length===1&&this.special_ids.has(n[0])){if(this.decode_kwargs.skip_special_tokens)return;if(this.token_cache.length>0){let e=this.tokenizer.decode(this.token_cache,this.decode_kwargs).slice(this.print_len);this.on_finalized_text(e,!1),this.token_cache=[],this.print_len=0}let e=this.tokenizer.decode(n,this.decode_kwargs);this.on_finalized_text(e,!1);return}this.token_cache=ci(this.token_cache,n);let r=this.tokenizer.decode(this.token_cache,this.decode_kwargs),i;t||r.endsWith(`
|
|
38
38
|
`)?(i=r.slice(this.print_len),this.token_cache=[],this.print_len=0):r.length>0&&wk(r.charCodeAt(r.length-1))?(i=r.slice(this.print_len),this.print_len+=i.length):(i=r.slice(this.print_len,r.lastIndexOf(` `)+1),this.print_len+=i.length),this.on_finalized_text(i,!1)}end(){let e;this.token_cache.length>0?(e=this.tokenizer.decode(this.token_cache,this.decode_kwargs).slice(this.print_len),this.token_cache=[],this.print_len=0):e=``,this.next_tokens_are_prompt=!0,this.on_finalized_text(e,!0)}on_finalized_text(e,t){e.length>0&&this.callback_function?.(e),t&&this.callback_function===Ek&&j.IS_PROCESS_AVAILABLE&&this.callback_function?.(`
|
|
39
|
-
`)}};Object.keys(Zc);export{RO as AutoModelForCausalLM,xm as AutoProcessor,G as AutoTokenizer,$r as LogLevel,Dk as TextStreamer,M as env};
|
|
39
|
+
`)}};Object.keys(Zc);export{RO as AutoModelForCausalLM,xm as AutoProcessor,G as AutoTokenizer,$r as LogLevel,Ud as RawImage,Dk as TextStreamer,M as env};
|