{"id": "e28796b1-3d10-57de-9d10-72aeb230afd8", "revision": 0, "last_node_id": 26, "last_link_id": 40, "nodes": [{"id": 1, "type": "VAELoader", "title": "H3 video VAE", "pos": [0, 0], "size": [390, 124], "flags": {}, "order": 0, "mode": 0, "inputs": [], "outputs": [{"name": "VAE", "type": "VAE", "links": [3, 20, 37]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VAELoader"}, "widgets_values": ["minimax_h3_video_vae_fp16.safetensors"]}, {"id": 2, "type": "VAELoader", "title": "H3 audio VAE", "pos": [0, 300], "size": [390, 124], "flags": {}, "order": 1, "mode": 0, "inputs": [], "outputs": [{"name": "VAE", "type": "VAE", "links": [4, 21, 38]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VAELoader"}, "widgets_values": ["minimax_h3_audio_vae_fp32.safetensors"]}, {"id": 3, "type": "CLIPLoader", "title": "H3 CLIP", "pos": [0, 600], "size": [390, 212], "flags": {}, "order": 2, "mode": 0, "inputs": [], "outputs": [{"name": "CLIP", "type": "CLIP", "links": [2, 19]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "CLIPLoader"}, "widgets_values": ["qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", "minimax", "default"]}, {"id": 4, "type": "UNETLoader", "title": "H3 FL2VA", "pos": [0, 900], "size": [390, 168], "flags": {}, "order": 3, "mode": 0, "inputs": [], "outputs": [{"name": "MODEL", "type": "MODEL", "links": [1]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "UNETLoader"}, "widgets_values": ["minimax_h3_fl2va_pruned_int8_convrot.safetensors", "default"]}, {"id": 5, "type": "LoraLoaderBypassModelOnly", "title": "LightX2V FL2V Turbo · corrected alpha8 · quantized-model bypass", "pos": [440, 0], "size": [390, 194], "flags": {}, "order": 4, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 1}], "outputs": [{"name": "MODEL", "type": "MODEL", "links": [6, 26]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "LoraLoaderBypassModelOnly"}, "widgets_values": ["minimax_h3_fl2v_turbo_4step_v0.1_comfyui_alpha8.safetensors", 1.0]}, {"id": 6, "type": "LoadImage", "title": "I2VA/Hybrid first frame", "pos": [0, 1200], "size": [390, 124], "flags": {}, "order": 5, "mode": 0, "inputs": [], "outputs": [{"name": "IMAGE", "type": "IMAGE", "links": [5, 22]}, {"name": "MASK", "type": "MASK", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "LoadImage"}, "widgets_values": ["10A.jpg"]}, {"id": 7, "type": "MiniMaxH3AudioConditioningT8", "title": "LOW Conditioning · I2VA native speech", "pos": [440, 300], "size": [390, 712], "flags": {}, "order": 6, "mode": 0, "inputs": [{"name": "clip", "type": "CLIP", "link": 2}, {"name": "video_vae", "type": "VAE", "link": 3}, {"name": "audio_vae", "type": "VAE", "link": 4}, {"name": "drive_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "final_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "first_frame", "type": "IMAGE", "link": 5, "shape": 7}, {"name": "last_frame", "type": "IMAGE", "link": null, "shape": 7}], "outputs": [{"name": "positive", "type": "CONDITIONING", "links": [10]}, {"name": "av_latent", "type": "LATENT", "links": [7, 15]}, {"name": "mux_audio", "type": "AUDIO", "links": null}, {"name": "conditioned_prompt", "type": "STRING", "links": null}, {"name": "media_map_json", "type": "STRING", "links": null}, {"name": "report", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AudioConditioningT8"}, "widgets_values": ["Locked-off medium close-up of the woman facing camera. She says clearly: <d>你在干嘛呢，我在这里呀，看看效果如何。</d> Natural, precise synchronized lip movements. Speak the complete Chinese sentence exactly once, with no added mumbling or repeated words. Minimal head motion, no music, no subtitles.", 736, 416, 124, "I2VA", "native", 1.0, false, 0, true, "match", "official_2_to_15s", false]}, {"id": 8, "type": "MiniMaxH3DualClockSamplerT8", "title": "LOW upstream I2V sampler contract · shift 12/3", "pos": [880, 0], "size": [390, 352], "flags": {}, "order": 7, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 6}, {"name": "av_latent", "type": "LATENT", "link": 7}], "outputs": [{"name": "model", "type": "MODEL", "links": [8, 9]}, {"name": "sampler", "type": "SAMPLER", "links": [13]}, {"name": "sigmas", "type": "SIGMAS", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3DualClockSamplerT8"}, "widgets_values": [8, 12.0, 3.0, "dual_clock_euler", "native_flow"]}, {"id": 9, "type": "MiniMaxH3LearnedTwoPassParityPlanT8Advanced", "title": "Published LBH schedule · simple8 split4 + refine4 · 8 total NFE", "pos": [1320, 0], "size": [390, 238], "flags": {}, "order": 8, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 8}], "outputs": [{"name": "coarse_sigmas", "type": "SIGMAS", "links": [14]}, {"name": "refine_sigmas", "type": "SIGMAS", "links": [28]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3LearnedTwoPassParityPlanT8Advanced"}, "widgets_values": [8, 4, 4]}, {"id": 10, "type": "BasicGuider", "title": "LOW guider", "pos": [1320, 300], "size": [390, 132], "flags": {}, "order": 9, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 9}, {"name": "conditioning", "type": "CONDITIONING", "link": 10}], "outputs": [{"name": "GUIDER", "type": "GUIDER", "links": [12]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "BasicGuider"}, "widgets_values": []}, {"id": 11, "type": "RandomNoise", "title": "LOW noise", "pos": [0, 1500], "size": [390, 142], "flags": {}, "order": 10, "mode": 0, "inputs": [], "outputs": [{"name": "NOISE", "type": "NOISE", "links": [11]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "RandomNoise"}, "widgets_values": [2608215001, "fixed"]}, {"id": 12, "type": "SamplerCustomAdvanced", "title": "PASS 1 - consume denoised_output", "pos": [1760, 0], "size": [390, 210], "flags": {}, "order": 11, "mode": 0, "inputs": [{"name": "noise", "type": "NOISE", "link": 11}, {"name": "guider", "type": "GUIDER", "link": 12}, {"name": "sampler", "type": "SAMPLER", "link": 13}, {"name": "sigmas", "type": "SIGMAS", "link": 14}, {"name": "latent_image", "type": "LATENT", "link": 15}], "outputs": [{"name": "output", "type": "LATENT", "links": null}, {"name": "denoised_output", "type": "LATENT", "links": [16]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "SamplerCustomAdvanced"}, "widgets_values": []}, {"id": 13, "type": "MiniMaxH3LearnedLatentUpscaleT8Advanced", "title": "Upstream-default learned 3D latent upscale · 2x", "pos": [2200, 0], "size": [390, 546], "flags": {}, "order": 12, "mode": 0, "inputs": [{"name": "av_latent", "type": "LATENT", "link": 16}], "outputs": [{"name": "av_latent", "type": "LATENT", "links": [23]}, {"name": "width", "type": "INT", "links": [17]}, {"name": "height", "type": "INT", "links": [18]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3LearnedLatentUpscaleT8Advanced"}, "widgets_values": ["minimax_h3_latent_upscaler_3d_fp16.safetensors", "scale_by", 2.0, 1.0, 1280, 704, "preserve_source", 1.05, "fp16", "offload_after"]}, {"id": 14, "type": "MiniMaxH3AudioConditioningT8", "title": "HIGH Conditioning · I2VA native speech", "pos": [2640, 0], "size": [390, 676], "flags": {}, "order": 13, "mode": 0, "inputs": [{"name": "clip", "type": "CLIP", "link": 19}, {"name": "video_vae", "type": "VAE", "link": 20}, {"name": "audio_vae", "type": "VAE", "link": 21}, {"name": "width", "type": "INT", "link": 17, "widget": {"name": "width"}}, {"name": "height", "type": "INT", "link": 18, "widget": {"name": "height"}}, {"name": "drive_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "final_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "first_frame", "type": "IMAGE", "link": 22, "shape": 7}, {"name": "last_frame", "type": "IMAGE", "link": null, "shape": 7}], "outputs": [{"name": "positive", "type": "CONDITIONING", "links": [25]}, {"name": "av_latent", "type": "LATENT", "links": [24]}, {"name": "mux_audio", "type": "AUDIO", "links": null}, {"name": "conditioned_prompt", "type": "STRING", "links": null}, {"name": "media_map_json", "type": "STRING", "links": null}, {"name": "report", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AudioConditioningT8"}, "widgets_values": ["Locked-off medium close-up of the woman facing camera. She says clearly: <d>你在干嘛呢，我在这里呀，看看效果如何。</d> Natural, precise synchronized lip movements. Speak the complete Chinese sentence exactly once, with no added mumbling or repeated words. Minimal head motion, no music, no subtitles.", 1344, 768, 124, "I2VA", "native", 1.0, false, 0, true, "match", "official_2_to_15s", true]}, {"id": 15, "type": "MiniMaxH3TwoPassLatentReconcileT8Advanced", "title": "Validate HIGH contract · pass 2 owns final AV", "pos": [3080, 0], "size": [390, 262], "flags": {}, "order": 14, "mode": 0, "inputs": [{"name": "learned_latent", "type": "LATENT", "link": 23}, {"name": "highres_template", "type": "LATENT", "link": 24}, {"name": "positive", "type": "CONDITIONING", "link": 25}], "outputs": [{"name": "av_latent", "type": "LATENT", "links": [27, 35]}, {"name": "positive", "type": "CONDITIONING", "links": [30]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3TwoPassLatentReconcileT8Advanced"}, "widgets_values": ["auto", "legacy_policy", 0.0]}, {"id": 16, "type": "MiniMaxH3TwoPassDetailMixerT8Advanced", "title": "HIGH upstream refine · optional Tail/Bias/STG/Restart off by default", "pos": [3520, 0], "size": [390, 994], "flags": {}, "order": 15, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 26}, {"name": "av_latent", "type": "LATENT", "link": 27}, {"name": "refine_sigmas", "type": "SIGMAS", "link": 28}], "outputs": [{"name": "model", "type": "MODEL", "links": [29]}, {"name": "sampler", "type": "SAMPLER", "links": [33]}, {"name": "sigmas", "type": "SIGMAS", "links": [34]}, {"name": "actual_nfe", "type": "INT", "links": null}, {"name": "planned_joint_av_forwards", "type": "INT", "links": null}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3TwoPassDetailMixerT8Advanced"}, "widgets_values": [12.0, 3.0, false, 3, "video_sigma_linear", false, -0.025, 0.7, 0.95, "video_sigma", false, 0.35, "25", 0.25, 0.85, false, 0.15, 3, 2608215001]}, {"id": 17, "type": "BasicGuider", "title": "HIGH guider", "pos": [3960, 0], "size": [390, 132], "flags": {}, "order": 16, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 29}, {"name": "conditioning", "type": "CONDITIONING", "link": 30}], "outputs": [{"name": "GUIDER", "type": "GUIDER", "links": [32]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "BasicGuider"}, "widgets_values": []}, {"id": 18, "type": "RandomNoise", "title": "HIGH restart noise", "pos": [0, 1800], "size": [390, 142], "flags": {}, "order": 17, "mode": 0, "inputs": [], "outputs": [{"name": "NOISE", "type": "NOISE", "links": [31]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "RandomNoise"}, "widgets_values": [2608215001, "fixed"]}, {"id": 19, "type": "SamplerCustomAdvanced", "title": "PASS 2 · decode output, not denoised_output", "pos": [4400, 0], "size": [390, 210], "flags": {}, "order": 18, "mode": 0, "inputs": [{"name": "noise", "type": "NOISE", "link": 31}, {"name": "guider", "type": "GUIDER", "link": 32}, {"name": "sampler", "type": "SAMPLER", "link": 33}, {"name": "sigmas", "type": "SIGMAS", "link": 34}, {"name": "latent_image", "type": "LATENT", "link": 35}], "outputs": [{"name": "output", "type": "LATENT", "links": [36]}, {"name": "denoised_output", "type": "LATENT", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "SamplerCustomAdvanced"}, "widgets_values": []}, {"id": 20, "type": "MiniMaxH3AVDecodeT8", "title": "Decode HIGH AV", "pos": [5280, 0], "size": [390, 158], "flags": {}, "order": 20, "mode": 0, "inputs": [{"name": "av_latent", "type": "LATENT", "link": 36}, {"name": "video_vae", "type": "VAE", "link": 37}, {"name": "audio_vae", "type": "VAE", "link": 38}], "outputs": [{"name": "frames", "type": "IMAGE", "links": [39]}, {"name": "generated_audio", "type": "AUDIO", "links": [40]}, {"name": "video_latent", "type": "LATENT", "links": null}, {"name": "audio_latent", "type": "LATENT", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AVDecodeT8"}, "widgets_values": []}, {"id": 21, "type": "VHS_VideoCombine", "title": "Save synchronized MP4 · generated AV audio", "pos": [5720, 0], "size": [390, 396], "flags": {}, "order": 21, "mode": 0, "inputs": [{"name": "images", "type": "IMAGE", "link": 39}, {"name": "audio", "type": "AUDIO", "link": 40, "shape": 7}, {"name": "meta_batch", "type": "VHS_BatchManager", "link": null, "shape": 7}, {"name": "vae", "type": "VAE", "link": null, "shape": 7}], "outputs": [{"name": "Filenames", "type": "VHS_FILENAMES", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VHS_VideoCombine"}, "widgets_values": [24, 0, "MiniMaxH3/learned_twopass_i2va_native_speech", "video/h265-mp4", false, true]}, {"id": 22, "type": "MarkdownNote", "title": "1 · I2VA native speech · current 4+4 route", "pos": [0, -440], "size": [520, 280], "flags": {}, "order": 22, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 1 · I2VA native speech / 当前4+4路线\nThis workflow uses the 4+4 learned two-pass route: LOW 736x416x124, 2x learned video-latent upscale, HIGH 1472x832x124, shift 12/3, and seed 2608215001. The fourth refine call inserts the published video sigma 0.8 instead of adding a tail step. Replace the first image and, where present, the loaded audio for your own material."]}, {"id": 23, "type": "MarkdownNote", "title": "2 · Automatic size synchronization", "pos": [540, -440], "size": [520, 280], "flags": {}, "order": 23, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 2 · Automatic size synchronization / 尺寸自动同步\nChange only Learned Latent Upscale `scale_by`. Its width/height outputs feed HIGH Conditioning. Do not maintain a second manual canvas."]}, {"id": 24, "type": "MarkdownNote", "title": "3 · Audio ownership and final connection", "pos": [1080, -440], "size": [520, 280], "flags": {}, "order": 24, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 3 · Audio ownership / 音频归属\nNo drive/reference audio is connected. LOW and HIGH Conditioning both use native audio generation, pass 2 remains unmasked, and Save receives generated AV Decode audio."]}, {"id": 25, "type": "MarkdownNote", "title": "4 · Keep optional detail controls off", "pos": [1620, -440], "size": [520, 280], "flags": {}, "order": 25, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 4 · Optional detail controls / 可选细节控制\nKeep Tail, model-time Bias, STG, and Restart OFF to preserve the reviewed route. Enabling any of them changes joint AV prediction and requires a fresh full audio/video review."]}, {"id": 26, "type": "MarkdownNote", "title": "5 · Validation scope after 4+4 update", "pos": [2160, -440], "size": [520, 280], "flags": {}, "order": 26, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 5 · Validation scope / 验证边界\nThe standard 4+4 Mandarin native-speech graph completed real HEVC/AAC generation, strict decode, and ASR intelligibility checks. The older per-audio-mode review used 4+3; this file now uses the corrected 4+4 schedule but this exact mode/seed was not re-rendered in the current round. Listen again before making a mode-specific quality claim. Nothing here proves universal lip-sync, voice identity, quality, or 16GB safety."]}], "links": [[1, 4, 0, 5, 0, "MODEL"], [2, 3, 0, 7, 0, "CLIP"], [3, 1, 0, 7, 1, "VAE"], [4, 2, 0, 7, 2, "VAE"], [5, 6, 0, 7, 5, "IMAGE"], [6, 5, 0, 8, 0, "MODEL"], [7, 7, 1, 8, 1, "LATENT"], [8, 8, 0, 9, 0, "MODEL"], [9, 8, 0, 10, 0, "MODEL"], [10, 7, 0, 10, 1, "CONDITIONING"], [11, 11, 0, 12, 0, "NOISE"], [12, 10, 0, 12, 1, "GUIDER"], [13, 8, 1, 12, 2, "SAMPLER"], [14, 9, 0, 12, 3, "SIGMAS"], [15, 7, 1, 12, 4, "LATENT"], [16, 12, 1, 13, 0, "LATENT"], [17, 13, 1, 14, 3, "INT"], [18, 13, 2, 14, 4, "INT"], [19, 3, 0, 14, 0, "CLIP"], [20, 1, 0, 14, 1, "VAE"], [21, 2, 0, 14, 2, "VAE"], [22, 6, 0, 14, 7, "IMAGE"], [23, 13, 0, 15, 0, "LATENT"], [24, 14, 1, 15, 1, "LATENT"], [25, 14, 0, 15, 2, "CONDITIONING"], [26, 5, 0, 16, 0, "MODEL"], [27, 15, 0, 16, 1, "LATENT"], [28, 9, 1, 16, 2, "SIGMAS"], [29, 16, 0, 17, 0, "MODEL"], [30, 15, 1, 17, 1, "CONDITIONING"], [31, 18, 0, 19, 0, "NOISE"], [32, 17, 0, 19, 1, "GUIDER"], [33, 16, 1, 19, 2, "SAMPLER"], [34, 16, 2, 19, 3, "SIGMAS"], [35, 15, 0, 19, 4, "LATENT"], [36, 19, 0, 20, 0, "LATENT"], [37, 1, 0, 20, 1, "VAE"], [38, 2, 0, 20, 2, "VAE"], [39, 20, 0, 21, 0, "IMAGE"], [40, 20, 1, 21, 1, "AUDIO"]], "groups": [], "config": {}, "extra": {"ds": {"scale": 0.75, "offset": [120, 120]}, "workflow_title": "MiniMax H3 learned latent two-pass I2VA · validated 4+4 standard"}, "version": 0.4}