完整指南演示如何用 ComfyUI 作为无头后端,搭建自动化的 MiniMax-H3 多模态推理环境,含硬件 profile、模型下载、动态图构建与联合解码。
本教程我们将使用 ComfyUI 作为无头推理后端,实现一个端到端的 MiniMax-H3 视频生成工作流。我们围绕 GPU 显存、磁盘容量、模型精度、分辨率、时长、采样策略及多种生成模式来配置环境,同时根据可用硬件动态选择合适的权重配置档。我们在程序中安装并启动 ComfyUI,从 Hugging Face 下载所需的 diffusion、text-encoder、video-VAE 和 audio-VAE 权重,并通过其 HTTP 和 WebSocket API 与运行中的服务器通信。我们还直接在 Python 中构建 ComfyUI 执行图,针对实时的 /object_info 端点验证节点 schema,支持文生视频(text-to-video)、首帧/尾帧条件生成,以及参考图条件生成。通过结合自动化模型安装、schema 感知图构建、视频音频联合解码、进度监控和输出收集,我们创建了一个可复现的流水线,无需依赖图形化 ComfyUI 界面即可实验 MiniMax-H3。
import json, os, re, shutil, subprocess, sys, time, uuid, urllib.request, urllib.error
from pathlib import Path
CFG = {
"MODE": "t2v",
"PROMPT": (
"Realistic live-action cinematic look. A lone lighthouse keeper on a storm-lashed "
"cliff at dusk, anamorphic lens, shallow depth of field, film grain, volumetric sea spray.\n"
"[0s-2s] Wide shot: waves detonate against black rock, the lighthouse beam sweeps the frame.\n"
"[2s-4s] Medium shot: the keeper braces against the wind, coat snapping, rain on his face.\n"
"[4s-5s] Close up: he squints into the dark and says \"She's holding.\"\n"
"Camera: hard cuts between shots, slight handheld jitter, no dissolves.\n"
"Audio: roaring surf and howling wind throughout, low cello drone underneath, "
"a heavy wave impact on each cut, the line delivered clearly over the storm.\n"
"No text, subtitles, logos or watermarks."
),
"ASPECT": (16, 9),
"MEGAPIXELS": 0.4,
"SECONDS": 5.0,
"SEED": 556589502035082,
"STEPS": 20,
"SAMPLER": "res_multistep",
"SCHEDULER": "simple",
"FIRST_FRAME": None,
"LAST_FRAME": None,
"REF_IMAGES": [],
"REF_IMAGE_SIZE": "match",
"SIGMA_SHIFT": None,
"TURBO_LORA": False,
"TURBO_STEPS": 8,
"TURBO_SAMPLER": "euler",
"TURBO_SCHEDULER": "beta",
"COMFY_DIR": "/content/ComfyUI",
"OUT_DIR": "/content/outputs",
"MODELS_ROOT": "/content/models",
"PORT": 8188,
"HF_TOKEN": os.environ.get("HF_TOKEN", ""),
"SKIP_INSTALL": False,
}
REPO = "Comfy-Org/MiniMax-H3"
API = f"http://127.0.0.1:{CFG['PORT']}"
PROFILES = [
dict(name="quality", min_vram=70,
unet_fl="minimax_h3_fl2va_bf16.safetensors",
unet_ref="minimax_h3_ref2va_bf16.safetensors",
te="qwen3vl_32b_minimax_h3_int8_convrot.safetensors",
flags=["--normalvram"]),
dict(name="balanced", min_vram=38,
unet_fl="minimax_h3_fl2va_pruned_int8_convrot.safetensors",
unet_ref="minimax_h3_ref2va_pruned_int8_convrot.safetensors",
te="qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors",
flags=["--normalvram", "--cache-none"]),
dict(name="squeeze", min_vram=20,
unet_fl="minimax_h3_fl2va_pruned_fp8_scaled.safetensors",
unet_ref="minimax_h3_ref2va_pruned_fp8_scaled.safetensors",
te="qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors",
flags=["--lowvram", "--cache-none", "--disable-smart-memory"]),
]
VAE_VIDEO = "minimax_h3_video_vae_fp16.safetensors"
VAE_AUDIO = "minimax_h3_audio_vae_fp32.safetensors"
def sh(cmd, cwd=None, check=True, quiet=False):
"""Run a shell command, streaming output."""
print(f"$ {cmd}")
p = subprocess.run(cmd, shell=True, cwd=cwd,
stdout=subprocess.DEVNULL if quiet else None,
stderr=subprocess.STDOUT if quiet else None)
if check and p.returncode != 0:
raise RuntimeError(f"command failed ({p.returncode}): {cmd}")
def get_json(path, payload=None, timeout=30):
url = f"{API}{path}"
data = json.dumps(payload).encode() if payload is not None else None
req = urllib.request.Request(url, data=data,
headers={"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=timeout) as r:
body = r.read()
return json.loads(body) if body else {}
def align_frames(seconds, fps=24):
"""H3 consumes frame counts on the 17k+5 grid. Snap upward."""
n = max(5, int(round(seconds * fps)))
while n % 17 != 5:
n += 1
return n
def h3_canvas(aspect=(16, 9), megapixels=0.98, multiple=32):
"""Mirror of ComfyUI's ResolutionSelector + H3's 768*1344 area cap."""
ar = aspect[0] / aspect[1]
total = megapixels * 1e6
h = (total / ar) ** 0.5
w = ar * h
cap = 768 * 1344
if w * h > cap:
我们定义了整个工作流中使用的核心 MiniMax-H3 配置、模型配置档、生成参数和共享工具函数。我们在推理开始前计算有效帧数和画布尺寸,同时检查 GPU 性能、可用 VRAM、BF16 支持和磁盘空间。我们还自动选择最合适的模型配置档,使流水线与 Colab 运行环境中的可用硬件相匹配。
def install_comfy():
comfy = Path(CFG["COMFY_DIR"])
if CFG["SKIP_INSTALL"] and comfy.exists():
print("Skipping install (SKIP_INSTALL=True).")
return
sh("pip install -q -U 'huggingface_hub[hf_xet]' hf_transfer websocket-client")
if not comfy.exists():
sh(f"git clone --depth 1 https://github.com/comfyanonymous/ComfyUI {comfy}")
sh(f"pip install -q -r {comfy}/requirements.txt")
ver = (comfy / "comfyui_version.py")
if ver.exists():
print("ComfyUI:", ver.read_text().strip())
if not (comfy / "comfy_extras" / "nodes_minimax_h3.py").exists():
raise SystemExit("This ComfyUI checkout lacks native MiniMax-H3 nodes — update it.")
root = Path(CFG["MODELS_ROOT"])
for sub in ("diffusion_models", "text_encoders", "vae", "loras"):
(root / sub).mkdir(parents=True, exist_ok=True)
(comfy / "extra_model_paths.yaml").write_text(
"minimax_h3:\n"
f" base_path: {root}\n"
" diffusion_models: diffusion_models\n"
" text_encoders: text_encoders\n"
" vae: vae\n"
" loras: loras\n"
)
Path(CFG["OUT_DIR"]).mkdir(parents=True, exist_ok=True)
def fetch(repo_id, filename, subdir):
from huggingface_hub import hf_hub_download
os.environ["HF_HUB_ENABLE_HF_TRANSFER"] = "1"
dest = Path(CFG["MODELS_ROOT"]) / subdir
target = dest / Path(filename).name
if target.exists() and target.stat().st_size > 1_000_000:
print(f"cached {target.name} ({target.stat().st_size/1e9:.1f} GB)")
return target
print(f"pulling {filename} -> {dest}")
try:
p = hf_hub_download(repo_id=repo_id, filename=filename, local_dir=str(dest),
token=CFG["HF_TOKEN"] or None)
except Exception as e:
if "401" in str(e) or "403" in str(e) or "gated" in str(e).lower():
raise SystemExit(
f"Access denied for {repo_id}. Accept the MiniMax-H3 community license on the "
"model page, create a read token, then set CFG['HF_TOKEN']."
) from e
raise
p = Path(p)
if p != target:
target.parent.mkdir(parents=True, exist_ok=True)
shutil.move(str(p), str(target))
return target
def download_weights(profile, mode):
unet = profile["unet_ref"] if mode == "r2v" else profile["unet_fl"]
fetch(REPO, f"diffusion_models/{unet}", "diffusion_models")
fetch(REPO, f"text_encoders/{profile['te']}", "text_encoders")
fetch(REPO, f"vae/{VAE_VIDEO}", "vae")
fetch(REPO, f"vae/{VAE_AUDIO}", "vae")
lora = None
if CFG["TURBO_LORA"]:
from huggingface_hub import HfApi
lora_repo = "drbaph/MiniMax-H3-Turbo-Lora-ComfyUI"
files = [f for f in HfApi().list_repo_files(lora_repo)
if f.endswith(".safetensors") and "pruned" in f]
if not files:
files = [f for f in HfApi().list_repo_files(lora_repo) if f.endswith(".safetensors")]
if files:
lora = fetch(lora_repo, sorted(files)[-1], "loras").name
print(f"turbo LoRA: {lora}")
return unet, profile["te"], lora
We install and configure ComfyUI, prepare the external model directory structure, and enable MiniMax-H3 support inside the Colab environment. We download the required diffusion model, text encoder, video VAE, and audio VAE weights from Hugging Face while reusing cached files whenever possible. We also optionally retrieve the Turbo LoRA configuration, allowing us to trade some generation quality for faster inference when required.
我们安装并配置 ComfyUI,准备外部模型目录结构,并在 Colab 环境中启用 MiniMax-H3 支持。我们从 Hugging Face 下载所需的扩散模型、文本编码器、视频 VAE 和音频 VAE 权重,同时尽可能复用缓存文件。我们还可选地获取 Turbo LoRA 配置,以便在需要时牺牲部分生成质量来换取更快的推理速度。
class ComfyServer:
def __init__(self, flags):
self.flags, self.proc, self.log = flags, None, Path("/content/comfyui.log")
def start(self):
cmd = [sys.executable, "main.py",
"--listen", "127.0.0.1", "--port", str(CFG["PORT"]),
"--disable-auto-launch", "--preview-method", "none",
"--output-directory", CFG["OUT_DIR"]] + self.flags
print("$", " ".join(cmd))
f = open(self.log, "wb")
self.proc = subprocess.Popen(cmd, cwd=CFG["COMFY_DIR"], stdout=f, stderr=subprocess.STDOUT)
deadline = time.time() + 300
while time.time() < deadline:
if self.proc.poll() is not None:
print(self.log.read_text()[-4000:])
raise SystemExit("ComfyUI died during startup (log above).")
try:
stats = get_json("/system_stats", timeout=3)
dev = stats.get("devices", [{}])[0]
print(f"server up — {dev.get('name','?')} "
f"{dev.get('vram_total',0)/1e9:.1f} GB total, "
f"{dev.get('vram_free',0)/1e9:.1f} GB free")
return
except Exception:
time.sleep(2)
raise SystemExit("Server did not become ready in 300s. Check /content/comfyui.log")
def tail(self, n=3000):
return self.log.read_text()[-n:] if self.log.exists() else ""
def free_vram(self):
try:
get_json("/free", {"unload_models": True, "free_memory": True})
except Exception:
pass
def stop(self):
if self.proc and self.proc.poll() is None:
self.proc.terminate()
try:
self.proc.wait(30)
except subprocess.TimeoutExpired:
self.proc.kill()
class Schema:
"""Reads /object_info so the graph is validated against the *running* node set
instead of whatever the docs said last week."""
def __init__(self):
self.info = get_json("/object_info", timeout=120)
def require(self, *classes):
missing = [c for c in classes if c not in self.info]
if missing:
raise SystemExit(f"Missing node classes: {missing}. Update ComfyUI to >= 0.30.0.")
def inputs_of(self, cls):
spec = self.info[cls]["input"]
return list(spec.get("required", {})) + list(spec.get("optional", {}))
def check(self, cls, payload):
known = set(self.inputs_of(cls))
unknown = [k for k in payload if k not in known]
if unknown:
print(f" note: {cls} does not declare {unknown} — declared: {sorted(known)}")
def autogrow(self, cls, prefix, n):
"""Autogrow slots (ref_image_1, ref_video_1, ...) are dynamic; discover the
real names if the server exposes them, otherwise fall back to 1-based."""
found = sorted([k for k in self.inputs_of(cls) if k.startswith(prefix)])
if len(found) >= n:
return found[:n]
return [f"{prefix}{i+1}" for i in range(n)]
We create a server-management layer that launches ComfyUI as a background subprocess and verifies that it becomes available through its API. We monitor server startup, inspect GPU memory statistics, free VRAM when necessary, and safely terminate the server after execution. We also build a schema-inspection utility that reads live ComfyUI node definitions so we can validate graph inputs and dynamically discover supported node slots.
我们创建了一个服务器管理层,将 ComfyUI 作为后台子进程启动,并通过其 API 验证其可用性。我们监控服务器启动过程、检查 GPU 内存统计信息、在需要时释放 VRAM,并在执行后安全地终止服务器。我们还构建了一个模式检查工具,用于读取实时的 ComfyUI 节点定义,从而可以验证图输入并动态发现支持的节点槽位。
class H3Graph:
def __init__(self, schema, unet, te, lora=None):
self.s, self.g, self._id = schema, {}, 0
self.unet, self.te, self.lora = unet, te, lora
def node(self, cls, **inputs):
self.s.check(cls, inputs)
self._id += 1
nid = str(self._id)
self.g[nid] = {"class_type": cls, "inputs": inputs}
return nid
def _backbone(self):
model = self.node("UNETLoader", unet_name=self.unet, weight_dtype="default")
if self.lora:
model = self.node("LoraLoaderModelOnly", model=[model, 0],
lora_name=self.lora, strength_model=1.0)
if CFG["SIGMA_SHIFT"]:
sv, sa = CFG["SIGMA_SHIFT"]
model = self.node("MiniMaxH3SigmaShift", model=[model, 0],
shift_video=float(sv), shift_audio=float(sa))
clip = self.node("CLIPLoader", clip_name=self.te, type="minimax", device="default")
vvae = self.node("VAELoader", vae_name=VAE_VIDEO)
avae = self.node("VAELoader", vae_name=VAE_AUDIO)
return model, clip, vvae, avae
def _tail(self, model, cond, latent, vvae, avae):
turbo = bool(self.lora)
steps = CFG["TURBO_STEPS"] if turbo else CFG["STEPS"]
sampler_name = CFG["TURBO_SAMPLER"] if turbo else CFG["SAMPLER"]
sched = CFG["TURBO_SCHEDULER"] if turbo else CFG["SCHEDULER"]
noise = self.node("RandomNoise", noise_seed=int(CFG["SEED"]))
samp = self.node("KSamplerSelect", sampler_name=sampler_name)
sig = self.node("BasicScheduler", model=[model, 0], scheduler=sched,
steps=steps, denoise=1.0)
guider = self.node("BasicGuider", model=[model, 0], conditioning=[cond[0], cond[1]])
out = self.node("SamplerCustomAdvanced", noise=[noise, 0], guider=[guider, 0],
sampler=[samp, 0], sigmas=[sig, 0], latent_image=[latent[0], latent[1]])
frames = self.node("VAEDecode", samples=[out, 0], vae=[vvae, 0])
audio = self.node("VAEDecodeAudio", samples=[out, 0], vae=[avae, 0])
vid = self.node("CreateVideo", images=[frames, 0], audio=[audio, 0], fps=24)
self.node("SaveVideo", video=[vid, 0], filename_prefix="MiniMaxH3/h3",
format="auto", codec="auto")
print(f" sampling: {steps} steps, {sampler_name}/{sched}")
return self.g
def _load_image(self, uploaded_name):
return self.node("LoadImage", image=uploaded_name, upload="image")
def t2v_or_flf2v(self, w, h, length, first=None, last=None):
self.s.require("MiniMaxH3ImageToVideo", "SamplerCustomAdvanced", "SaveVideo")
model, clip, vvae, avae = self._backbone()
kw = {}
if first:
kw["first_frame"] = [self._load_image(first), 0]
if last:
kw["last_frame"] = [self._load_image(last), 0]
n = self.node("MiniMaxH3ImageToVideo", clip=[clip, 0], vae=[vvae, 0],
prompt=CFG["PROMPT"], width=w, height=h, length=length, **kw)
return self._tail(model, (n, 0), (n, 1), vvae, avae)
def r2v(self, w, h, length, ref_names):
self.s.require("MiniMaxH3ReferenceToVideo")
model, clip, vvae, avae = self._backbone()
slots = self.s.autogrow("MiniMaxH3ReferenceToVideo", "ref_image_", len(ref_names))
refs = {slot: [self._load_image(nm), 0] for slot, nm in zip(slots, ref_names)}
print(f" reference slots: {list(refs)}")
n = self.node("MiniMaxH3ReferenceToVideo", clip=[clip, 0], vae=[vvae, 0],
audio_vae=[avae, 0], prompt=CFG["PROMPT"], width=w, height=h,
length=length, ref_image_size=CFG["REF_IMAGE_SIZE"], **refs)
return self._tail(model, (n, 0), (n, 1), vvae, avae)
We construct the MiniMax-H3 ComfyUI workflow graph entirely in Python using reusable node-building methods. We assemble the model backbone, conditioning pipeline, sampler, schedulers, joint latent decoding, video creation, and output-saving stages for both standard and Turbo configurations. We also support text-to-video, first- and last-frame-conditioned video, and reference-image-conditioned video generation through the same programmable graph architecture.
我们使用可复用的节点构建方法,完全在 Python 中构建 MiniMax-H3 ComfyUI 工作流图。我们为标准配置和 Turbo 配置组装了模型主干、条件管线、采样器、调度器、联合潜在解码、视频创建和输出保存阶段。我们还通过同一个可编程图架构支持文生视频、首尾帧条件视频以及参考图像条件视频的生成。
def upload_image(path):
"""Multipart POST to /upload/image; returns the name LoadImage expects."""
path = Path(path)
if not path.exists():
raise FileNotFoundError(path)
boundary = uuid.uuid4().hex
body = (
f"--{boundary}\r\nContent-Disposition: form-data; name=\"image\"; "
f"filename=\"{path.name}\"\r\nContent-Type: application/octet-stream\r\n\r\n"
).encode() + path.read_bytes() + (
f"\r\n--{boundary}\r\nContent-Disposition: form-data; name=\"overwrite\"\r\n\r\ntrue"
f"\r\n--{boundary}--\r\n"
).encode()
req = urllib.request.Request(f"{API}/upload/image", data=body,
headers={"Content-Type": f"multipart/form-data; boundary={boundary}"})
with urllib.request.urlopen(req, timeout=120) as r:
info = json.loads(r.read())
sub = info.get("subfolder") or ""
print(f" uploaded {path.name}")
return f"{sub}/{info['name']}" if sub else info["name"]
def run_graph(graph, server, timeout=7200):
"""Submit, then follow the WebSocket for per-step progress."""
import websocket
cid = uuid.uuid4().hex
Path("/content/last_workflow_api.json").write_text(json.dumps(graph, indent=2))
try:
res = get_json("/prompt", {"prompt": graph, "client_id": cid})
except urllib.error.HTTPError as e:
detail = e.read().decode()[:3000]
raise SystemExit(f"Graph rejected by ComfyUI:\n{detail}")
pid = res["prompt_id"]
print(f"queued {pid} — first run loads ~37 GB of weights, be patient")
ws = websocket.WebSocket()
ws.connect(f"ws://127.0.0.1:{CFG['PORT']}/ws?clientId={cid}", timeout=60)
t0, last = time.time(), ""
try:
while time.time() - t0 < timeout:
try:
msg = ws.recv()
except Exception:
time.sleep(1)
continue
if isinstance(msg, bytes):
continue
d = json.loads(msg)
t, data = d.get("type"), d.get("data", {})
if t == "executing" and data.get("prompt_id") == pid:
if data.get("node") is None:
print(f"\ndone in {time.time()-t0:.0f}s")
break
cls = graph.get(data["node"], {}).get("class_type", data["node"])
if cls != last:
print(f"\n -> {cls}", end="", flush=True)
last = cls
elif t == "progress":
v, m = data.get("value", 0), data.get("max", 1)
print(f"\r -> {last} {v}/{m} ", end="", flush=True)
elif t == "execution_error":
print("\n--- execution error ---")
print(json.dumps(data, indent=2)[:4000])
print(server.tail())
raise SystemExit("Generation failed.")
finally:
ws.close()
files = []
try:
hist = get_json(f"/history/{pid}")
for out in hist.get(pid, {}).get("outputs", {}).values():
for items in out.values():
if isinstance(items, list):
for it in items:
if isinstance(it, dict) and "filename" in it:
p = Path(CFG["OUT_DIR"]) / (it.get("subfolder") or "") / it["filename"]
if p.exists():
files.append(p)
except Exception:
pass
if not files:
cands = [p for p in Path(CFG["OUT_DIR"]).rglob("*")
if p.suffix.lower() in (".mp4", ".webm", ".mkv") and p.stat().st_mtime > t0]
files = sorted(cands, key=lambda p: p.stat().st_mtime)
return files
def main():
profile = preflight()
install_comfy()
mode = CFG["MODE"]
unet, te, lora = download_weights(profile, mode)
w, h = h3_canvas(CFG["ASPECT"], CFG["MEGAPIXELS"])
length = align_frames(CFG["SECONDS"])
print(f"\ncanvas {w}x{h}, {length} frames "
f"({length/24:.2f}s @24fps, grid check {length % 17 == 5})")
server = ComfyServer(profile["flags"])
server.start()
我们处理图片上传、图提交、WebSocket 进度跟踪、输出发现以及教程的完整执行流程。我们将生成的图提交给 ComfyUI,监控各个节点的执行和采样进度,收集生成的视频文件,并在 Colab 内部直接展示可管理的输出。最终,我们通过 main 函数协调所有早期组件,将工作流从硬件预检和模型加载推进到同步的 MiniMax-H3 视频和音频生成。
总之,我们实现了一条完整的可编程 MiniMax-H3 推理流水线,从硬件验证和模型获取到图执行,最终实现同步的视频-音频生成。我们使用 ComfyUI 作为无头服务器,同时从 Python 端控制整个工作流,这让我们能够直接访问配置、模型加载、条件设置、采样、解码、服务器生命周期管理以及生成输出。我们还通过动态检查 ComfyUI 节点 schema、根据可用 VRAM 调整模型配置、使帧数与 MiniMax-H3 要求对齐,以及通过同一可复用架构支持多种条件模式,使流水线更加健壮。在工作流结束时,我们拥有了一个灵活的基础,可以在此基础上使用不同的提示词、种子、参考图片、帧约束、LoRA 加速、分辨率和采样策略进行扩展,同时保持一致且自动化的 MiniMax-H3 生成流程。
在此处获取完整代码。另外,欢迎关注我们的 Twitter,记得订阅我们的 Newsletter。你用 telegram 吗?现在你也可以加入我们的 telegram 群了。
需要与我们合作推广你的 GitHub 仓库或 Hugging Face 页面或产品发布或网络研讨会等?请与我们联系。
Sana Hassan,Marktechpost 咨询实习生,同时是印度理工学院马德拉斯分校的双学位学生,热衷于将技术和 AI 应用于解决现实世界的挑战。凭借解决实际问题的浓厚兴趣,他为 AI 与现实生活交汇处带来了全新的视角。