{"id": "dd9a75ac-056e-5d60-9556-c0c1dd7dad05", "revision": 0, "last_node_id": 27, "last_link_id": 42, "nodes": [{"id": 1, "type": "VAELoader", "title": "H3 video VAE", "pos": [0, 0], "size": [390, 124], "flags": {}, "order": 0, "mode": 0, "inputs": [], "outputs": [{"name": "VAE", "type": "VAE", "links": [3, 20, 37]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VAELoader"}, "widgets_values": ["minimax_h3_video_vae_fp16.safetensors"]}, {"id": 2, "type": "VAELoader", "title": "H3 audio VAE", "pos": [0, 300], "size": [390, 124], "flags": {}, "order": 1, "mode": 0, "inputs": [], "outputs": [{"name": "VAE", "type": "VAE", "links": [4, 21, 38]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VAELoader"}, "widgets_values": ["minimax_h3_audio_vae_fp32.safetensors"]}, {"id": 3, "type": "CLIPLoader", "title": "H3 CLIP", "pos": [0, 600], "size": [390, 212], "flags": {}, "order": 2, "mode": 0, "inputs": [], "outputs": [{"name": "CLIP", "type": "CLIP", "links": [2, 19]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "CLIPLoader"}, "widgets_values": ["qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", "minimax", "default"]}, {"id": 4, "type": "UNETLoader", "title": "H3 FL2VA", "pos": [0, 900], "size": [390, 168], "flags": {}, "order": 3, "mode": 0, "inputs": [], "outputs": [{"name": "MODEL", "type": "MODEL", "links": [1]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "UNETLoader"}, "widgets_values": ["minimax_h3_fl2va_pruned_int8_convrot.safetensors", "default"]}, {"id": 5, "type": "LoraLoaderBypassModelOnly", "title": "LightX2V FL2V Turbo · corrected alpha8 · quantized-model bypass", "pos": [440, 0], "size": [390, 194], "flags": {}, "order": 4, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 1}], "outputs": [{"name": "MODEL", "type": "MODEL", "links": [6, 26]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "LoraLoaderBypassModelOnly"}, "widgets_values": ["minimax_h3_fl2v_turbo_4step_v0.1_comfyui_alpha8.safetensors", 1.0]}, {"id": 6, "type": "LoadImage", "title": "I2VA/Hybrid first frame", "pos": [0, 1200], "size": [390, 124], "flags": {}, "order": 5, "mode": 0, "inputs": [], "outputs": [{"name": "IMAGE", "type": "IMAGE", "links": [5, 22]}, {"name": "MASK", "type": "MASK", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "LoadImage"}, "widgets_values": ["10A.jpg"]}, {"id": 7, "type": "MiniMaxH3AudioConditioningT8", "title": "LOW Conditioning · Hybrid reference_only", "pos": [440, 300], "size": [390, 712], "flags": {}, "order": 6, "mode": 0, "inputs": [{"name": "clip", "type": "CLIP", "link": 2}, {"name": "video_vae", "type": "VAE", "link": 3}, {"name": "audio_vae", "type": "VAE", "link": 4}, {"name": "drive_audio", "type": "AUDIO", "link": 41, "shape": 7}, {"name": "final_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "first_frame", "type": "IMAGE", "link": 5, "shape": 7}, {"name": "last_frame", "type": "IMAGE", "link": null, "shape": 7}], "outputs": [{"name": "positive", "type": "CONDITIONING", "links": [10]}, {"name": "av_latent", "type": "LATENT", "links": [7, 15]}, {"name": "mux_audio", "type": "AUDIO", "links": null}, {"name": "conditioned_prompt", "type": "STRING", "links": null}, {"name": "media_map_json", "type": "STRING", "links": null}, {"name": "report", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AudioConditioningT8"}, "widgets_values": ["Locked-off medium close-up of the woman facing camera. Use <Audio 1> only as rhythm and pacing reference and generate a new voice saying exactly: <d>All the time he was talking to me, his angry little eyes were following Lake.</d> Keep visible lip movements precisely synchronized to the generated speech. Minimal head motion, no music, no subtitles.", 736, 416, 124, "Hybrid", "reference_only", 1.0, true, 1, true, "match", "official_2_to_15s", false]}, {"id": 8, "type": "MiniMaxH3DualClockSamplerT8", "title": "LOW upstream I2V sampler contract · shift 12/3", "pos": [880, 0], "size": [390, 352], "flags": {}, "order": 7, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 6}, {"name": "av_latent", "type": "LATENT", "link": 7}], "outputs": [{"name": "model", "type": "MODEL", "links": [8, 9]}, {"name": "sampler", "type": "SAMPLER", "links": [13]}, {"name": "sigmas", "type": "SIGMAS", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3DualClockSamplerT8"}, "widgets_values": [8, 12.0, 3.0, "dual_clock_euler", "native_flow"]}, {"id": 9, "type": "MiniMaxH3LearnedTwoPassParityPlanT8Advanced", "title": "Published LBH schedule · simple8 split4 + refine4 · 8 total NFE", "pos": [1320, 0], "size": [390, 238], "flags": {}, "order": 8, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 8}], "outputs": [{"name": "coarse_sigmas", "type": "SIGMAS", "links": [14]}, {"name": "refine_sigmas", "type": "SIGMAS", "links": [28]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3LearnedTwoPassParityPlanT8Advanced"}, "widgets_values": [8, 4, 4]}, {"id": 10, "type": "BasicGuider", "title": "LOW guider", "pos": [1320, 300], "size": [390, 132], "flags": {}, "order": 9, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 9}, {"name": "conditioning", "type": "CONDITIONING", "link": 10}], "outputs": [{"name": "GUIDER", "type": "GUIDER", "links": [12]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "BasicGuider"}, "widgets_values": []}, {"id": 11, "type": "RandomNoise", "title": "LOW noise", "pos": [0, 1500], "size": [390, 142], "flags": {}, "order": 10, "mode": 0, "inputs": [], "outputs": [{"name": "NOISE", "type": "NOISE", "links": [11]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "RandomNoise"}, "widgets_values": [2608215001, "fixed"]}, {"id": 12, "type": "SamplerCustomAdvanced", "title": "PASS 1 - consume denoised_output", "pos": [1760, 0], "size": [390, 210], "flags": {}, "order": 11, "mode": 0, "inputs": [{"name": "noise", "type": "NOISE", "link": 11}, {"name": "guider", "type": "GUIDER", "link": 12}, {"name": "sampler", "type": "SAMPLER", "link": 13}, {"name": "sigmas", "type": "SIGMAS", "link": 14}, {"name": "latent_image", "type": "LATENT", "link": 15}], "outputs": [{"name": "output", "type": "LATENT", "links": null}, {"name": "denoised_output", "type": "LATENT", "links": [16]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "SamplerCustomAdvanced"}, "widgets_values": []}, {"id": 13, "type": "MiniMaxH3LearnedLatentUpscaleT8Advanced", "title": "Upstream-default learned 3D latent upscale · 2x", "pos": [2200, 0], "size": [390, 546], "flags": {}, "order": 12, "mode": 0, "inputs": [{"name": "av_latent", "type": "LATENT", "link": 16}], "outputs": [{"name": "av_latent", "type": "LATENT", "links": [23]}, {"name": "width", "type": "INT", "links": [17]}, {"name": "height", "type": "INT", "links": [18]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3LearnedLatentUpscaleT8Advanced"}, "widgets_values": ["minimax_h3_latent_upscaler_3d_fp16.safetensors", "scale_by", 2.0, 1.0, 1280, 704, "preserve_source", 1.05, "fp16", "offload_after"]}, {"id": 14, "type": "MiniMaxH3AudioConditioningT8", "title": "HIGH Conditioning · Hybrid reference_only", "pos": [2640, 0], "size": [390, 676], "flags": {}, "order": 13, "mode": 0, "inputs": [{"name": "clip", "type": "CLIP", "link": 19}, {"name": "video_vae", "type": "VAE", "link": 20}, {"name": "audio_vae", "type": "VAE", "link": 21}, {"name": "width", "type": "INT", "link": 17, "widget": {"name": "width"}}, {"name": "height", "type": "INT", "link": 18, "widget": {"name": "height"}}, {"name": "drive_audio", "type": "AUDIO", "link": 42, "shape": 7}, {"name": "final_audio", "type": "AUDIO", "link": null, "shape": 7}, {"name": "first_frame", "type": "IMAGE", "link": 22, "shape": 7}, {"name": "last_frame", "type": "IMAGE", "link": null, "shape": 7}], "outputs": [{"name": "positive", "type": "CONDITIONING", "links": [25]}, {"name": "av_latent", "type": "LATENT", "links": [24]}, {"name": "mux_audio", "type": "AUDIO", "links": null}, {"name": "conditioned_prompt", "type": "STRING", "links": null}, {"name": "media_map_json", "type": "STRING", "links": null}, {"name": "report", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AudioConditioningT8"}, "widgets_values": ["Locked-off medium close-up of the woman facing camera. Use <Audio 1> only as rhythm and pacing reference and generate a new voice saying exactly: <d>All the time he was talking to me, his angry little eyes were following Lake.</d> Keep visible lip movements precisely synchronized to the generated speech. Minimal head motion, no music, no subtitles.", 1344, 768, 124, "Hybrid", "reference_only", 1.0, true, 1, true, "match", "official_2_to_15s", true]}, {"id": 15, "type": "MiniMaxH3TwoPassLatentReconcileT8Advanced", "title": "Validate HIGH contract · pass 2 owns final AV", "pos": [3080, 0], "size": [390, 262], "flags": {}, "order": 14, "mode": 0, "inputs": [{"name": "learned_latent", "type": "LATENT", "link": 23}, {"name": "highres_template", "type": "LATENT", "link": 24}, {"name": "positive", "type": "CONDITIONING", "link": 25}], "outputs": [{"name": "av_latent", "type": "LATENT", "links": [27, 35]}, {"name": "positive", "type": "CONDITIONING", "links": [30]}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3TwoPassLatentReconcileT8Advanced"}, "widgets_values": ["auto", "legacy_policy", 0.0]}, {"id": 16, "type": "MiniMaxH3TwoPassDetailMixerT8Advanced", "title": "HIGH upstream refine · optional Tail/Bias/STG/Restart off by default", "pos": [3520, 0], "size": [390, 994], "flags": {}, "order": 15, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 26}, {"name": "av_latent", "type": "LATENT", "link": 27}, {"name": "refine_sigmas", "type": "SIGMAS", "link": 28}], "outputs": [{"name": "model", "type": "MODEL", "links": [29]}, {"name": "sampler", "type": "SAMPLER", "links": [33]}, {"name": "sigmas", "type": "SIGMAS", "links": [34]}, {"name": "actual_nfe", "type": "INT", "links": null}, {"name": "planned_joint_av_forwards", "type": "INT", "links": null}, {"name": "report_json", "type": "STRING", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3TwoPassDetailMixerT8Advanced"}, "widgets_values": [12.0, 3.0, false, 3, "video_sigma_linear", false, -0.025, 0.7, 0.95, "video_sigma", false, 0.35, "25", 0.25, 0.85, false, 0.15, 3, 2608215001]}, {"id": 17, "type": "BasicGuider", "title": "HIGH guider", "pos": [3960, 0], "size": [390, 132], "flags": {}, "order": 16, "mode": 0, "inputs": [{"name": "model", "type": "MODEL", "link": 29}, {"name": "conditioning", "type": "CONDITIONING", "link": 30}], "outputs": [{"name": "GUIDER", "type": "GUIDER", "links": [32]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "BasicGuider"}, "widgets_values": []}, {"id": 18, "type": "RandomNoise", "title": "HIGH restart noise", "pos": [0, 1800], "size": [390, 142], "flags": {}, "order": 17, "mode": 0, "inputs": [], "outputs": [{"name": "NOISE", "type": "NOISE", "links": [31]}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "RandomNoise"}, "widgets_values": [2608215001, "fixed"]}, {"id": 19, "type": "SamplerCustomAdvanced", "title": "PASS 2 · decode output, not denoised_output", "pos": [4400, 0], "size": [390, 210], "flags": {}, "order": 18, "mode": 0, "inputs": [{"name": "noise", "type": "NOISE", "link": 31}, {"name": "guider", "type": "GUIDER", "link": 32}, {"name": "sampler", "type": "SAMPLER", "link": 33}, {"name": "sigmas", "type": "SIGMAS", "link": 34}, {"name": "latent_image", "type": "LATENT", "link": 35}], "outputs": [{"name": "output", "type": "LATENT", "links": [36]}, {"name": "denoised_output", "type": "LATENT", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "SamplerCustomAdvanced"}, "widgets_values": []}, {"id": 20, "type": "MiniMaxH3AVDecodeT8", "title": "Decode HIGH AV", "pos": [5280, 0], "size": [390, 158], "flags": {}, "order": 20, "mode": 0, "inputs": [{"name": "av_latent", "type": "LATENT", "link": 36}, {"name": "video_vae", "type": "VAE", "link": 37}, {"name": "audio_vae", "type": "VAE", "link": 38}], "outputs": [{"name": "frames", "type": "IMAGE", "links": [39]}, {"name": "generated_audio", "type": "AUDIO", "links": [40]}, {"name": "video_latent", "type": "LATENT", "links": null}, {"name": "audio_latent", "type": "LATENT", "links": null}], "properties": {"cnr_id": "minimax-h3-audio-T8", "Node name for S&R": "MiniMaxH3AVDecodeT8"}, "widgets_values": []}, {"id": 21, "type": "VHS_VideoCombine", "title": "Save synchronized MP4 · generated AV audio", "pos": [5720, 0], "size": [390, 396], "flags": {}, "order": 21, "mode": 0, "inputs": [{"name": "images", "type": "IMAGE", "link": 39}, {"name": "audio", "type": "AUDIO", "link": 40, "shape": 7}, {"name": "meta_batch", "type": "VHS_BatchManager", "link": null, "shape": 7}, {"name": "vae", "type": "VAE", "link": null, "shape": 7}], "outputs": [{"name": "Filenames", "type": "VHS_FILENAMES", "links": null}], "properties": {"cnr_id": "comfy-core", "Node name for S&R": "VHS_VideoCombine"}, "widgets_values": [24, 0, "MiniMaxH3/learned_twopass_hybrid_reference_only_speech", "video/h265-mp4", false, true]}, {"id": 22, "type": "MarkdownNote", "title": "1 · Hybrid reference_only · current 4+4 route", "pos": [0, -440], "size": [520, 280], "flags": {}, "order": 22, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 1 · Hybrid reference_only / 当前4+4路线\nThis workflow uses the 4+4 learned two-pass route: LOW 736x416x124, 2x learned video-latent upscale, HIGH 1472x832x124, shift 12/3, and seed 2608215001. The fourth refine call inserts the published video sigma 0.8 instead of adding a tail step. Replace the first image and, where present, the loaded audio for your own material."]}, {"id": 23, "type": "MarkdownNote", "title": "2 · Automatic size synchronization", "pos": [540, -440], "size": [520, 280], "flags": {}, "order": 23, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 2 · Automatic size synchronization / 尺寸自动同步\nChange only Learned Latent Upscale `scale_by`. Its width/height outputs feed HIGH Conditioning. Do not maintain a second manual canvas."]}, {"id": 24, "type": "MarkdownNote", "title": "3 · Audio ownership and final connection", "pos": [1080, -440], "size": [520, 280], "flags": {}, "order": 24, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 3 · Audio ownership / 音频归属\nThe loaded source is registered as `<Audio 1>` reference, while the target audio latent starts blank and is regenerated. Save AV Decode audio. This is not equivalent to `remix_source=1`; the strength widget is not the remix control in this mode."]}, {"id": 25, "type": "MarkdownNote", "title": "4 · Keep optional detail controls off", "pos": [1620, -440], "size": [520, 280], "flags": {}, "order": 25, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 4 · Optional detail controls / 可选细节控制\nKeep Tail, model-time Bias, STG, and Restart OFF to preserve the reviewed route. Enabling any of them changes joint AV prediction and requires a fresh full audio/video review."]}, {"id": 26, "type": "MarkdownNote", "title": "5 · Validation scope after 4+4 update", "pos": [2160, -440], "size": [520, 280], "flags": {}, "order": 26, "mode": 0, "inputs": [], "outputs": [], "properties": {"Node name for S&R": "MarkdownNote"}, "widgets_values": ["## 5 · Validation scope / 验证边界\nThe standard 4+4 Mandarin native-speech graph completed real HEVC/AAC generation, strict decode, and ASR intelligibility checks. The older per-audio-mode review used 4+3; this file now uses the corrected 4+4 schedule but this exact mode/seed was not re-rendered in the current round. Listen again before making a mode-specific quality claim. Nothing here proves universal lip-sync, voice identity, quality, or 16GB safety."]}, {"id": 27, "type": "LoadAudio", "title": "Drive/reference audio · replace as needed", "pos": [0, 1360], "size": [390, 120], "flags": {}, "order": 27, "mode": 0, "inputs": [], "outputs": [{"name": "AUDIO", "type": "AUDIO", "links": [41, 42]}], "properties": {"Node name for S&R": "LoadAudio"}, "widgets_values": ["h3_twopass_voice_5683_5p152s.flac"]}], "links": [[1, 4, 0, 5, 0, "MODEL"], [2, 3, 0, 7, 0, "CLIP"], [3, 1, 0, 7, 1, "VAE"], [4, 2, 0, 7, 2, "VAE"], [5, 6, 0, 7, 5, "IMAGE"], [6, 5, 0, 8, 0, "MODEL"], [7, 7, 1, 8, 1, "LATENT"], [8, 8, 0, 9, 0, "MODEL"], [9, 8, 0, 10, 0, "MODEL"], [10, 7, 0, 10, 1, "CONDITIONING"], [11, 11, 0, 12, 0, "NOISE"], [12, 10, 0, 12, 1, "GUIDER"], [13, 8, 1, 12, 2, "SAMPLER"], [14, 9, 0, 12, 3, "SIGMAS"], [15, 7, 1, 12, 4, "LATENT"], [16, 12, 1, 13, 0, "LATENT"], [17, 13, 1, 14, 3, "INT"], [18, 13, 2, 14, 4, "INT"], [19, 3, 0, 14, 0, "CLIP"], [20, 1, 0, 14, 1, "VAE"], [21, 2, 0, 14, 2, "VAE"], [22, 6, 0, 14, 7, "IMAGE"], [23, 13, 0, 15, 0, "LATENT"], [24, 14, 1, 15, 1, "LATENT"], [25, 14, 0, 15, 2, "CONDITIONING"], [26, 5, 0, 16, 0, "MODEL"], [27, 15, 0, 16, 1, "LATENT"], [28, 9, 1, 16, 2, "SIGMAS"], [29, 16, 0, 17, 0, "MODEL"], [30, 15, 1, 17, 1, "CONDITIONING"], [31, 18, 0, 19, 0, "NOISE"], [32, 17, 0, 19, 1, "GUIDER"], [33, 16, 1, 19, 2, "SAMPLER"], [34, 16, 2, 19, 3, "SIGMAS"], [35, 15, 0, 19, 4, "LATENT"], [36, 19, 0, 20, 0, "LATENT"], [37, 1, 0, 20, 1, "VAE"], [38, 2, 0, 20, 2, "VAE"], [39, 20, 0, 21, 0, "IMAGE"], [40, 20, 1, 21, 1, "AUDIO"], [41, 27, 0, 7, 3, "AUDIO"], [42, 27, 0, 14, 5, "AUDIO"]], "groups": [], "config": {}, "extra": {"ds": {"scale": 0.75, "offset": [120, 120]}, "workflow_title": "MiniMax H3 learned latent two-pass I2VA · validated 4+4 standard"}, "version": 0.4}