AI News HubLIVE
In-site rewrite6 min read

Implementing a MiniMax-H3 Multimodal Video and Audio Generation Pipeline with ComfyUI APIs

In this comprehensive guide, we demonstrate how to implement a complete, programmable MiniMax-H3 multimodal generation pipeline. By leveraging ComfyUI as a headless backend, we walk through setting up an automated inference environment that handles hardware profiling, model weight downloading, dynamic graph construction, and joint video-audio decoding. The post Implementing a MiniMax-H3 Multimodal Video and Audio Generation Pipeline with ComfyUI APIs appeared first on MarkTechPost.

SourceMarkTechPostAuthor: Sana Hassan

In this tutorial, we implement an end-to-end MiniMax-H3 video generation workflow using ComfyUI as a headless inference backend. We configure the environment around GPU memory, disk capacity, model precision, resolution, duration, sampling strategy, and multiple generation modes, while dynamically selecting an appropriate weight profile based on the available hardware. We install and launch ComfyUI programmatically, download the required diffusion, text-encoder, video-VAE, and audio-VAE weights from Hugging Face, and communicate with the running server through its HTTP and WebSocket APIs. We also construct the ComfyUI execution graph directly in Python, validate node schemas against the live /object_info endpoint, and support text-to-video, first- and last-frame-conditioned generation, and reference-image-conditioned generation. By combining automated model setup, schema-aware graph construction, joint video-audio decoding, progress monitoring, and output collection, we create a reproducible pipeline for experimenting with MiniMax-H3 without relying on the graphical ComfyUI interface. Copy CodeCopiedUse a different Browser import json, os, re, shutil, subprocess, sys, time, uuid, urllib.request, urllib.error from pathlib import Path CFG = { "MODE": "t2v", "PROMPT": ( "Realistic live-action cinematic look. A lone lighthouse keeper on a storm-lashed " "cliff at dusk, anamorphic lens, shallow depth of field, film grain, volumetric sea spray.\n" "[0s-2s] Wide shot: waves detonate against black rock, the lighthouse beam sweeps the frame.\n" "[2s-4s] Medium shot: the keeper braces against the wind, coat snapping, rain on his face.\n" "[4s-5s] Close up: he squints into the dark and says \"She's holding.\"\n" "Camera: hard cuts between shots, slight handheld jitter, no dissolves.\n" "Audio: roaring surf and howling wind throughout, low cello drone underneath, " "a heavy wave impact on each cut, the line delivered clearly over the storm.\n" "No text, subtitles, logos or watermarks." ), "ASPECT": (16, 9), "MEGAPIXELS": 0.4, "SECONDS": 5.0, "SEED": 556589502035082, "STEPS": 20, "SAMPLER": "res_multistep", "SCHEDULER": "simple", "FIRST_FRAME": None, "LAST_FRAME": None, "REF_IMAGES": [], "REF_IMAGE_SIZE": "match", "SIGMA_SHIFT": None, "TURBO_LORA": False, "TURBO_STEPS": 8, "TURBO_SAMPLER": "euler", "TURBO_SCHEDULER": "beta", "COMFY_DIR": "/content/ComfyUI", "OUT_DIR": "/content/outputs", "MODELS_ROOT": "/content/models", "PORT": 8188, "HF_TOKEN": os.environ.get("HF_TOKEN", ""), "SKIP_INSTALL": False, } REPO = "Comfy-Org/MiniMax-H3" API = f"http://127.0.0.1:{CFG['PORT']}" PROFILES = [ dict(name="quality", min_vram=70, unet_fl="minimax_h3_fl2va_bf16.safetensors", unet_ref="minimax_h3_ref2va_bf16.safetensors", te="qwen3vl_32b_minimax_h3_int8_convrot.safetensors", flags=["--normalvram"]), dict(name="balanced", min_vram=38, unet_fl="minimax_h3_fl2va_pruned_int8_convrot.safetensors", unet_ref="minimax_h3_ref2va_pruned_int8_convrot.safetensors", te="qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", flags=["--normalvram", "--cache-none"]), dict(name="squeeze", min_vram=20, unet_fl="minimax_h3_fl2va_pruned_fp8_scaled.safetensors", unet_ref="minimax_h3_ref2va_pruned_fp8_scaled.safetensors", te="qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", flags=["--lowvram", "--cache-none", "--disable-smart-memory"]), ] VAE_VIDEO = "minimax_h3_video_vae_fp16.safetensors" VAE_AUDIO = "minimax_h3_audio_vae_fp32.safetensors" def sh(cmd, cwd=None, check=True, quiet=False): """Run a shell command, streaming output.""" print(f"$ {cmd}") p = subprocess.run(cmd, shell=True, cwd=cwd, stdout=subprocess.DEVNULL if quiet else None, stderr=subprocess.STDOUT if quiet else None) if check and p.returncode != 0: raise RuntimeError(f"command failed ({p.returncode}): {cmd}") def get_json(path, payload=None, timeout=30): url = f"{API}{path}" data = json.dumps(payload).encode() if payload is not None else None req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=timeout) as r: body = r.read() return json.loads(body) if body else {} def align_frames(seconds, fps=24): """H3 consumes frame counts on the 17k+5 grid. Snap upward.""" n = max(5, int(round(seconds * fps))) while n % 17 != 5: n += 1 return n def h3_canvas(aspect=(16, 9), megapixels=0.98, multiple=32): """Mirror of ComfyUI's ResolutionSelector + H3's 768*1344 area cap.""" ar = aspect[0] / aspect[1] total = megapixels * 1e6 h = (total / ar) 0.5 w = ar * h cap = 768 * 1344 if w * h > cap: s = (cap / (w * h)) 0.5 w, h = w * s, h * s r = lambda v: max(multiple, int(round(v / multiple)) * multiple) return r(w), r(h) def preflight(): try: import torch except ImportError: raise SystemExit("PyTorch missing — run this in a Colab GPU runtime.") if not torch.cuda.is_available(): raise SystemExit("No CUDA device. Runtime > Change runtime type > GPU (A100).") name = torch.cuda.get_device_name(0) vram = torch.cuda.get_device_properties(0).total_memory / 1e9 free_disk = shutil.disk_usage("/content").free / 1e9 bf16 = torch.cuda.is_bf16_supported() print(f"GPU : {name} ({vram:.1f} GB VRAM, bf16={bf16})") print(f"Free disk : {free_disk:.1f} GB") if not bf16: raise SystemExit( "This GPU has no bf16 support (T4/K80). MiniMax-H3 will not run here.\n" "Switch to an A100/L4/H100 runtime." ) profile = next((p for p in PROFILES if vram >= p["min_vram"]), None) if profile is None: raise SystemExit( f"{vram:.0f} GB VRAM is below the ~20 GB floor for the smallest H3 build." ) if free_disk 1_000_000: print(f"cached {target.name} ({target.stat().st_size/1e9:.1f} GB)") return target print(f"pulling {filename} -> {dest}") try: p = hf_hub_download(repo_id=repo_id, filename=filename, local_dir=str(dest), token=CFG["HF_TOKEN"] or None) except Exception as e: if "401" in str(e) or "403" in str(e) or "gated" in str(e).lower(): raise SystemExit( f"Access denied for {repo_id}. Accept the MiniMax-H3 community license on the " "model page, create a read token, then set CFG['HF_TOKEN']." ) from e raise p = Path(p) if p != target: target.parent.mkdir(parents=True, exist_ok=True) shutil.move(str(p), str(target)) return target def download_weights(profile, mode): unet = profile["unet_ref"] if mode == "r2v" else profile["unet_fl"] fetch(REPO, f"diffusion_models/{unet}", "diffusion_models") fetch(REPO, f"text_encoders/{profile['te']}", "text_encoders") fetch(REPO, f"vae/{VAE_VIDEO}", "vae") fetch(REPO, f"vae/{VAE_AUDIO}", "vae") lora = None if CFG["TURBO_LORA"]: from huggingface_hub import HfApi lora_repo = "drbaph/MiniMax-H3-Turbo-Lora-ComfyUI" files = [f for f in HfApi().list_repo_files(lora_repo) if f.endswith(".safetensors") and "pruned" in f] if not files: files = [f for f in HfApi().list_repo_files(lora_repo) if f.endswith(".safetensors")] if files: lora = fetch(lora_repo, sorted(files)[-1], "loras").name print(f"turbo LoRA: {lora}") return unet, profile["te"], lora We install and configure ComfyUI, prepare the external model directory structure, and enable MiniMax-H3 support inside the Colab environment. We download the required diffusion model, text encoder, video VAE, and audio VAE weights from Hugging Face while reusing cached files whenever possible. We also optionally retrieve the Turbo LoRA configuration, allowing us to trade some generation quality for faster inference when required. Copy CodeCopiedUse a different Browser class ComfyServer: def init(self, flags): self.flags, self.proc, self.log = flags, None, Path("/content/comfyui.log") def start(self): cmd = [sys.executable, "main.py", "--listen", "127.0.0.1", "--port", str(CFG["PORT"]), "--disable-auto-launch", "--preview-method", "none", "--output-directory", CFG["OUT_DIR"]] + self.flags print("$", " ".join(cmd)) f = open(self.log, "wb") self.proc = subprocess.Popen(cmd, cwd=CFG["COMFY_DIR"], stdout=f, stderr=subprocess.STDOUT) deadline = time.time() + 300 while time.time() = 0.30.0.") def inputs_of(self, cls): spec = self.info[cls]["input"] return list(spec.get("required", {})) + list(spec.get("optional", {})) def check(self, cls, payload): known = set(self.inputs_of(cls)) unknown = [k for k in payload if k not in known] if unknown: print(f" note: {cls} does not declare {unknown} — declared: {sorted(known)}") def autogrow(self, cls, prefix, n): """Autogrow slots (ref_image_1, ref_video_1, ...) are dynamic; discover the real names if the server exposes them, otherwise fall back to 1-based.""" found = sorted([k for k in self.inputs_of(cls) if k.startswith(prefix)]) if len(found) >= n: return found[:n] return [f"{prefix}{i+1}" for i in range(n)] We create a server-management layer that launches ComfyUI as a background subprocess and verifies that it becomes available through its API. We monitor server startup, inspect GPU memory statistics, free VRAM when necessary, and safely terminate the server after execution. We also build a schema-inspection utility that reads live ComfyUI node definitions so we can validate graph inputs and dynamically discover supported node slots. Copy CodeCopiedUse a different Browser class H3Graph: def init(self, schema, unet, te, lora=None): self.s, self.g, self._id = schema, {}, 0 self.unet, self.te, self.lora = unet, te, lora def node(self, cls, inputs): self.s.check(cls, inputs) self._id += 1 nid = str(self._id) self.g[nid] = {"class_type": cls, "inputs": inputs} return nid def _backbone(self): model = self.node("UNETLoader", unet_name=self.unet, weight_dtype="default") if self.lora: model = self.node("LoraLoaderModelOnly", model=[model, 0], lora_name=self.lora, strength_model=1.0) if CFG["SIGMA_SHIFT"]: sv, sa = CFG["SIGMA_SHIFT"] model = self.node("MiniMaxH3SigmaShift", model=[model, 0], shift_video=float(sv), shift_audio=float(sa)) clip = self.node("CLIPLoader", clip_name=self.te, type="minimax", device="default") vvae = self.node("VAELoader", vae_name=VAE_VIDEO) avae = self.node("VAELoader", vae_name=VAE_AUDIO) return model, clip, vvae, avae def _tail(self, model, cond, latent, vvae, avae): turbo = bool(self.lora) steps = CFG["TURBO_STEPS"] if turbo else CFG["STEPS"] sampler_name = CFG["TURBO_SAMPLER"] if turbo else CFG["SAMPLER"] sched = CFG["TURBO_SCHEDULER"] if turbo else CFG["SCHEDULER"] noise = self.node("RandomNoise", noise_seed=int(CFG["SEED"])) samp = self.node("KSamplerSelect", sampler_name=sampler_name) sig = self.node("BasicScheduler", model=[model, 0], scheduler=sched, steps=steps, denoise=1.0) guider = self.node("BasicGuider", model=[model, 0], conditioning=[cond[0], cond[1]]) out = self.node("SamplerCustomAdvanced", noise=[noise, 0], guider=[guider, 0], sampler=[samp, 0], sigmas=[sig, 0], latent_image=[latent[0], latent[1]]) frames = self.node("VAEDecode", samples=[out, 0], vae=[vvae, 0]) audio = self.node("VAEDecodeAudio", samples=[out, 0], vae=[avae, 0]) vid = self.node("CreateVideo", images=[frames, 0], audio=[audio, 0], fps=24) self.node("SaveVideo", video=[vid, 0], filename_prefix="MiniMaxH3/h3", format="auto", codec="auto") print(f" sampling: {steps} steps, {sampler_name}/{sched}") return self.g def _load_image(self, uploaded_name): return self.node("LoadImage", image=uploaded_name, upload="image") def t2v_or_flf2v(self, w, h, length, first=None, last=None): self.s.require("MiniMaxH3ImageToVideo", "SamplerCustomAdvanced", "SaveVideo") model, clip, vvae, avae = self._backbone() kw = {} if first: kw["first_frame"] = [self._load_image(first), 0] if last: kw["last_frame"] = [self._load_image(last), 0] n = self.node("MiniMaxH3ImageToVideo", clip=[clip, 0], vae=[vvae, 0], prompt=CFG["PROMPT"], width=w, height=h, length=length, kw) return self._tail(model, (n, 0), (n, 1), vvae, avae) def r2v(self, w, h, length, ref_names): self.s.require("MiniMaxH3ReferenceToVideo") model, clip, vvae, avae = self._backbone() slots = self.s.autogrow("MiniMaxH3ReferenceToVideo", "ref_image_", len(ref_names)) refs = {slot: [self._load_image(nm), 0] for slot, nm in zip(slots, ref_n [truncated for AI cost control]