diff --git a/README.md b/README.md index f4402f4..f90508e 100644 --- a/README.md +++ b/README.md @@ -1 +1,44 @@ -cloud-worker +# cloud-worker + +Tooling for running the **MiniMax H3** open-weights video model (native stereo +audio, 24 fps, 5–15 s, ~1 MP native) on RunPod GPU machines via ComfyUI. + +Current test rig: 1x B200 (180 GB) pod — connection details in `pod-info.txt`. + +## Layout + +- `scripts/pod/` — runs on the pod + - `install-minimax-h3.sh` — one-shot setup on a fresh pod (after `install-comfy.sh`) + - `download-weights.sh` — all weight families → `/workspace/minimax-h3/models` + - `download-weights-tmpfs.sh` — overflow files → `/dev/shm` (network-volume + quota workaround; ephemeral, re-run after pod restart) + - `extra_model_paths.yaml` — installed to `/ComfyUI/extra_model_paths.yaml` + - `start-comfyui.sh` — headless ComfyUI in tmux on 127.0.0.1:8188 + - `make-test-images.py` — synthetic reference/first-frame images → `/ComfyUI/input` +- `scripts/local/port-forward.sh` — persistent tunnel `localhost:8188` → pod + ComfyUI panel (auto-reconnects) +- `bench/minimax_bench.py` — API-based benchmark harness (t2v / i2v / ref2v × + duration × resolution × weight family); appends to `bench/results.csv` + + `bench/results.jsonl` +- `docs/WEIGHTS.md` — which weights are downloaded, where, and which family +- `docs/RESEARCH.md` — model/ComfyUI findings, running without ComfyUI + (SGLang / vLLM-Omni / diffusers), concurrency model, RunPod serverless +- `docs/BENCHMARKS.md` — measured results on the B200 + +## Quick start + +```bash +# tunnel to the ComfyUI panel (leave running) +scripts/local/port-forward.sh & +open http://localhost:8188 + +# one-off generation through the API +python3 bench/minimax_bench.py --task t2v --width 864 --height 480 --seconds 5 + +# smoke suite / full sweep +python3 bench/minimax_bench.py --suite quick +python3 bench/minimax_bench.py --suite sweep --prefix prunedint8_ +``` + +On the pod, ComfyUI runs in tmux session `comfyui` +(`tmux attach -t comfyui`), logs at `/workspace/minimax-h3/comfyui.log`. diff --git a/bench/minimax_bench.py b/bench/minimax_bench.py new file mode 100644 index 0000000..5d1a2b6 --- /dev/null +++ b/bench/minimax_bench.py @@ -0,0 +1,334 @@ +#!/usr/bin/env python3 +"""Benchmark MiniMax H3 video generation through the ComfyUI API. + +Builds API-format graphs programmatically (no workflow JSON needed) for the +three local H3 task types and times each generation: + + t2v MiniMaxH3ImageToVideo with no frames connected (pure text) + i2v MiniMaxH3ImageToVideo with a first_frame image + ref2v MiniMaxH3ReferenceToVideo with one reference image + +Model facts baked in (see docs/RESEARCH.md): + - frame length must satisfy n % 17 == 5; trained range 124-362 (~5-15 s @ 24fps) + - canvas: multiples of 32, native area <= 768*1344 (~1 MP) + - reference sampler config: res_multistep + simple scheduler, 20 steps, no CFG + +Usage examples (through the SSH tunnel, or on the pod with --host): + python bench/minimax_bench.py --suite quick + python bench/minimax_bench.py --suite sweep --model fl2va=minimax_h3_fl2va_pruned_int8_convrot.safetensors \ + --model ref2va=minimax_h3_ref2va_pruned_int8_convrot.safetensors + python bench/minimax_bench.py --task t2v --width 864 --height 480 --seconds 5 --steps 20 + +Results are appended to bench/results.csv (one row per run) and full metadata +to bench/results.jsonl. Output videos are saved on the ComfyUI side under +output/bench/. +""" +import argparse +import csv +import json +import sys +import threading +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path + +DEFAULT_HOST = "http://127.0.0.1:8188" + +TEXT_ENCODER = "qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors" +VIDEO_VAE = "minimax_h3_video_vae_fp16.safetensors" +AUDIO_VAE = "minimax_h3_audio_vae_fp32.safetensors" +FL2VA_DEFAULT = "minimax_h3_fl2va_pruned_int8_convrot.safetensors" +REF2VA_DEFAULT = "minimax_h3_ref2va_pruned_int8_convrot.safetensors" +BENCH_IMAGE = "bench_ref_1344x768.png" # created by scripts/pod/make-test-images.py + +T2V_PROMPT = ( + "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, " + "waves crashing on dark rocks below, seagulls circling, warm sunlight flares, " + "sound of surf and wind, gentle orchestral swell." +) +I2V_PROMPT = ( + "The camera slowly pushes into the city skyline as dusk settles, window lights " + "flickering on one by one, light wind, distant traffic hum and a soft synth pad." +) +REF2V_PROMPT = ( + "Use as the setting. A slow cinematic pan across the sunset city " + "skyline, clouds drifting, lights turning on in the towers, ambient city sounds." +) + + +def snap_length(seconds: float) -> int: + """Duration in seconds -> valid H3 frame count (24 fps, n % 17 == 5, snapped up).""" + n = max(5, round(seconds * 24)) + return n + (5 - (n % 17)) % 17 + + +def build_graph(task, model_file, prompt, width, height, length, steps, seed, + sampler="res_multistep", scheduler="simple", image=BENCH_IMAGE, + ref_image_size="match", filename_prefix="bench/run"): + g = { + "1": {"class_type": "UNETLoader", + "inputs": {"unet_name": model_file, "weight_dtype": "default"}}, + "2": {"class_type": "CLIPLoader", + "inputs": {"clip_name": TEXT_ENCODER, "type": "minimax", "device": "default"}}, + "3": {"class_type": "VAELoader", "inputs": {"vae_name": VIDEO_VAE}}, + "4": {"class_type": "VAELoader", "inputs": {"vae_name": AUDIO_VAE}}, + "6": {"class_type": "KSamplerSelect", "inputs": {"sampler_name": sampler}}, + "7": {"class_type": "BasicScheduler", + "inputs": {"model": ["1", 0], "scheduler": scheduler, "steps": steps, + "denoise": 1.0}}, + "8": {"class_type": "RandomNoise", "inputs": {"noise_seed": seed}}, + "9": {"class_type": "BasicGuider", + "inputs": {"model": ["1", 0], "conditioning": ["5", 0]}}, + "11": {"class_type": "SamplerCustomAdvanced", + "inputs": {"noise": ["8", 0], "guider": ["9", 0], "sampler": ["6", 0], + "sigmas": ["7", 0], "latent_image": ["5", 1]}}, + "12": {"class_type": "VAEDecode", "inputs": {"samples": ["11", 0], "vae": ["3", 0]}}, + "13": {"class_type": "VAEDecodeAudio", "inputs": {"samples": ["11", 0], "vae": ["4", 0]}}, + "14": {"class_type": "CreateVideo", + "inputs": {"images": ["12", 0], "audio": ["13", 0], "fps": 24, "bit_depth": 8}}, + "15": {"class_type": "SaveVideo", + "inputs": {"video": ["14", 0], "filename_prefix": filename_prefix, + "format": "auto", "codec": "auto"}}, + } + common = {"prompt": prompt, "width": width, "height": height, "length": length} + if task == "t2v": + g["5"] = {"class_type": "MiniMaxH3ImageToVideo", + "inputs": {"clip": ["2", 0], "vae": ["3", 0], **common}} + elif task == "i2v": + g["10"] = {"class_type": "LoadImage", "inputs": {"image": image}} + g["5"] = {"class_type": "MiniMaxH3ImageToVideo", + "inputs": {"clip": ["2", 0], "vae": ["3", 0], "first_frame": ["10", 0], + **common}} + elif task == "ref2v": + g["10"] = {"class_type": "LoadImage", "inputs": {"image": image}} + g["5"] = {"class_type": "MiniMaxH3ReferenceToVideo", + "inputs": {"clip": ["2", 0], "vae": ["3", 0], "audio_vae": ["4", 0], + "ref_image_size": ref_image_size, + "ref_images.ref_image_0": ["10", 0], **common}} + else: + raise ValueError(f"unknown task {task}") + return g + + +class Api: + def __init__(self, host): + self.host = host.rstrip("/") + self.client_id = str(uuid.uuid4()) + + def _req(self, path, data=None, timeout=30): + url = self.host + path + body = json.dumps(data).encode() if data is not None else None + req = urllib.request.Request(url, data=body, + headers={"Content-Type": "application/json"} if body else {}) + with urllib.request.urlopen(req, timeout=timeout) as r: + return json.loads(r.read() or "{}") + + def queue_prompt(self, graph): + return self._req("/prompt", {"prompt": graph, "client_id": self.client_id}) + + def history(self, prompt_id): + return self._req(f"/history/{prompt_id}") + + def vram_used(self): + try: + s = self._req("/system_stats", timeout=5) + d = s["devices"][0] + return d["vram_total"] - d["vram_free"] + except Exception: + return None + + +def run_one(api, cfg, poll=2.0, timeout=3600): + """Submit one benchmark run; returns result dict.""" + graph = build_graph(**{k: v for k, v in cfg.items() if k in ( + "task", "model_file", "prompt", "width", "height", "length", "steps", "seed", + "sampler", "scheduler", "image", "ref_image_size", "filename_prefix")}) + peak = {"vram": 0} + stop = threading.Event() + + def sample_vram(): + while not stop.is_set(): + v = api.vram_used() + if v: + peak["vram"] = max(peak["vram"], v) + stop.wait(3) + + t = threading.Thread(target=sample_vram, daemon=True) + t.start() + t0 = time.time() + try: + resp = api.queue_prompt(graph) + except urllib.error.HTTPError as e: + stop.set() + detail = e.read().decode()[:2000] + return {**cfg, "ok": False, "error": f"HTTP {e.code}: {detail}"} + prompt_id = resp["prompt_id"] + exec_start = exec_end = None + status_str = None + while time.time() - t0 < timeout: + time.sleep(poll) + h = api.history(prompt_id) + if prompt_id in h: + entry = h[prompt_id] + status_str = entry.get("status", {}).get("status_str") + for msg in entry.get("status", {}).get("messages", []): + if msg[0] == "execution_start": + exec_start = msg[1].get("timestamp") + if msg[0] in ("execution_success", "execution_error"): + exec_end = msg[1].get("timestamp") + if status_str in ("success", "error"): + break + stop.set() + wall = time.time() - t0 + exec_s = (exec_end - exec_start) / 1000 if exec_start and exec_end else None + outputs = [] + if status_str == "success": + for node_out in h[prompt_id].get("outputs", {}).values(): + for key in ("images", "video", "videos"): + for f in node_out.get(key, []): + outputs.append(f.get("filename")) + return {**cfg, "ok": status_str == "success", "status": status_str, + "wall_s": round(wall, 1), + "exec_s": round(exec_s, 1) if exec_s else None, + "peak_vram_gb": round(peak["vram"] / 1e9, 1) if peak["vram"] else None, + "outputs": outputs, + "error": None if status_str == "success" else json.dumps( + h.get(prompt_id, {}).get("status", {}))[:2000]} + + +def append_results(row, outdir): + outdir.mkdir(parents=True, exist_ok=True) + jl = outdir / "results.jsonl" + with jl.open("a") as f: + f.write(json.dumps({**row, "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())}) + "\n") + csv_path = outdir / "results.csv" + fields = ["ts", "label", "task", "model_file", "width", "height", "seconds", "length", + "steps", "sampler", "scheduler", "ref_image_size", "warm", "ok", "wall_s", + "exec_s", "peak_vram_gb", "outputs", "error"] + new = not csv_path.exists() + with csv_path.open("a", newline="") as f: + w = csv.DictWriter(f, fieldnames=fields, extrasaction="ignore") + if new: + w.writeheader() + w.writerow({**row, "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + "outputs": ";".join(row.get("outputs") or [])}) + + +def default_prompt(task): + return {"t2v": T2V_PROMPT, "i2v": I2V_PROMPT, "ref2v": REF2V_PROMPT}[task] + + +def make_cfg(task, model_file, width, height, seconds, steps, seed, label, warm, + sampler="res_multistep", scheduler="simple", ref_image_size="match"): + return { + "label": label, "task": task, "model_file": model_file, + "prompt": default_prompt(task), "width": width, "height": height, + "seconds": seconds, "length": snap_length(seconds), "steps": steps, + "seed": seed, "sampler": sampler, "scheduler": scheduler, + "image": BENCH_IMAGE, "ref_image_size": ref_image_size, "warm": warm, + "filename_prefix": f"bench/{label}", + } + + +def suite_quick(fl2va, ref2va, prefix=""): + """Small smoke suite: one tiny run per task type.""" + return [ + make_cfg("t2v", fl2va, 864, 480, 5, 20, 1, f"{prefix}quick_t2v", warm=False), + make_cfg("i2v", fl2va, 864, 480, 5, 20, 1, f"{prefix}quick_i2v", warm=True), + make_cfg("ref2v", ref2va, 864, 480, 5, 20, 1, f"{prefix}quick_ref2v", warm=False), + ] + + +def suite_sweep(fl2va, ref2va, prefix=""): + """Full sweep: tasks x durations x resolutions at fixed 20 steps. + + Resolution ladder (16:9, multiples of 32, up to the 1MP native cap): + 608x352 (0.2MP draft), 864x480 (0.4MP default), 1344x768 (1MP max) + Durations: 5 s (124 fr), 10 s (243 fr), 15 s (362 fr) - the trained range. + """ + resolutions = [(608, 352), (864, 480), (1344, 768)] + durations = [5, 10, 15] + cfgs = [] + warm_fl = False + for w, h in resolutions: + for sec in durations: + cfgs.append(make_cfg("t2v", fl2va, w, h, sec, 20, 1, + f"{prefix}sweep_t2v_{w}x{h}_{sec}s", warm=warm_fl)) + warm_fl = True + # i2v: model already warm; vary duration at default res, plus max res spot check + for sec in durations: + cfgs.append(make_cfg("i2v", fl2va, 864, 480, sec, 20, 1, + f"{prefix}sweep_i2v_864x480_{sec}s", warm=True)) + cfgs.append(make_cfg("i2v", fl2va, 1344, 768, 5, 20, 1, + f"{prefix}sweep_i2v_1344x768_5s", warm=True)) + # ref2v: separate weights -> first run is a model swap (cold) + warm_ref = False + for sec in durations: + cfgs.append(make_cfg("ref2v", ref2va, 864, 480, sec, 20, 1, + f"{prefix}sweep_ref2v_864x480_{sec}s", warm=warm_ref)) + warm_ref = True + cfgs.append(make_cfg("ref2v", ref2va, 1344, 768, 5, 20, 1, + f"{prefix}sweep_ref2v_1344x768_5s", warm=True)) + return cfgs + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--host", default=DEFAULT_HOST) + ap.add_argument("--suite", choices=["quick", "sweep"]) + ap.add_argument("--task", choices=["t2v", "i2v", "ref2v"]) + ap.add_argument("--model", action="append", default=[], + help="override model files, e.g. fl2va=file.safetensors (repeatable)") + ap.add_argument("--width", type=int, default=864) + ap.add_argument("--height", type=int, default=480) + ap.add_argument("--seconds", type=float, default=5) + ap.add_argument("--steps", type=int, default=20) + ap.add_argument("--seed", type=int, default=1) + ap.add_argument("--sampler", default="res_multistep") + ap.add_argument("--scheduler", default="simple") + ap.add_argument("--ref-image-size", default="match", choices=["match", "max"]) + ap.add_argument("--label", default=None) + ap.add_argument("--prefix", default="", help="label prefix for suites, e.g. prunedint8_") + ap.add_argument("--warm", action="store_true", help="mark run as warm (model preloaded)") + ap.add_argument("--outdir", default=str(Path(__file__).parent)) + args = ap.parse_args() + + models = {"fl2va": FL2VA_DEFAULT, "ref2va": REF2VA_DEFAULT} + for m in args.model: + k, v = m.split("=", 1) + models[k] = v + + api = Api(args.host) + if args.suite == "quick": + cfgs = suite_quick(models["fl2va"], models["ref2va"], args.prefix) + elif args.suite == "sweep": + cfgs = suite_sweep(models["fl2va"], models["ref2va"], args.prefix) + elif args.task: + model = models["ref2va"] if args.task == "ref2v" else models["fl2va"] + label = args.label or f"{args.task}_{args.width}x{args.height}_{args.seconds}s" + cfgs = [make_cfg(args.task, model, args.width, args.height, args.seconds, + args.steps, args.seed, label, args.warm, + args.sampler, args.scheduler, args.ref_image_size)] + else: + ap.error("need --suite or --task") + + outdir = Path(args.outdir) + for i, cfg in enumerate(cfgs): + print(f"[{i+1}/{len(cfgs)}] {cfg['label']}: {cfg['task']} {cfg['model_file']} " + f"{cfg['width']}x{cfg['height']} {cfg['seconds']}s ({cfg['length']}fr) " + f"{cfg['steps']} steps ...", flush=True) + row = run_one(api, cfg) + append_results(row, outdir) + if row["ok"]: + print(f" OK wall={row['wall_s']}s exec={row['exec_s']}s " + f"peak_vram={row['peak_vram_gb']}GB -> {row['outputs']}", flush=True) + else: + print(f" FAILED: {str(row.get('error'))[:400]}", flush=True) + print("done.") + + +if __name__ == "__main__": + main() diff --git a/bench/results.csv b/bench/results.csv new file mode 100644 index 0000000..e202224 --- /dev/null +++ b/bench/results.csv @@ -0,0 +1,6 @@ +ts,label,task,model_file,width,height,seconds,length,steps,sampler,scheduler,ref_image_size,warm,ok,wall_s,exec_s,peak_vram_gb,outputs,error +2026-08-06T04:19:55Z,smoke_t2v,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,False,True,80.8,80.1,42.0,smoke_t2v_00001_.mp4, +2026-08-06T04:20:48Z,smoke_i2v,i2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,True,True,37.9,36.3,45.0,smoke_i2v_00001_.mp4, +2026-08-06T04:22:41Z,smoke_ref2v,ref2v,minimax_h3_ref2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,False,True,113.3,111.4,67.0,smoke_ref2v_00001_.mp4, +2026-08-06T04:26:04Z,prunedint8_sweep_t2v_608x352_5s,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5,124,20,res_multistep,simple,match,False,True,10.9,9.6,64.4,prunedint8_sweep_t2v_608x352_5s_00001_.mp4, +2026-08-06T04:27:19Z,prunedint8_sweep_t2v_608x352_10s,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,10,243,20,res_multistep,simple,match,True,True,75.9,74.6,68.4,prunedint8_sweep_t2v_608x352_10s_00001_.mp4, diff --git a/bench/results.jsonl b/bench/results.jsonl new file mode 100644 index 0000000..d900d38 --- /dev/null +++ b/bench/results.jsonl @@ -0,0 +1,5 @@ +{"label": "smoke_t2v", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/smoke_t2v", "ok": true, "status": "success", "wall_s": 80.8, "exec_s": 80.1, "peak_vram_gb": 42.0, "outputs": ["smoke_t2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:19:55Z"} +{"label": "smoke_i2v", "task": "i2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "The camera slowly pushes into the city skyline as dusk settles, window lights flickering on one by one, light wind, distant traffic hum and a soft synth pad.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": true, "filename_prefix": "bench/smoke_i2v", "ok": true, "status": "success", "wall_s": 37.9, "exec_s": 36.3, "peak_vram_gb": 45.0, "outputs": ["smoke_i2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:20:48Z"} +{"label": "smoke_ref2v", "task": "ref2v", "model_file": "minimax_h3_ref2va_pruned_int8_convrot.safetensors", "prompt": "Use as the setting. A slow cinematic pan across the sunset city skyline, clouds drifting, lights turning on in the towers, ambient city sounds.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/smoke_ref2v", "ok": true, "status": "success", "wall_s": 113.3, "exec_s": 111.4, "peak_vram_gb": 67.0, "outputs": ["smoke_ref2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:22:41Z"} +{"label": "prunedint8_sweep_t2v_608x352_5s", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 5, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/prunedint8_sweep_t2v_608x352_5s", "ok": true, "status": "success", "wall_s": 10.9, "exec_s": 9.6, "peak_vram_gb": 64.4, "outputs": ["prunedint8_sweep_t2v_608x352_5s_00001_.mp4"], "error": null, "ts": "2026-08-06T04:26:04Z"} +{"label": "prunedint8_sweep_t2v_608x352_10s", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 10, "length": 243, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": true, "filename_prefix": "bench/prunedint8_sweep_t2v_608x352_10s", "ok": true, "status": "success", "wall_s": 75.9, "exec_s": 74.6, "peak_vram_gb": 68.4, "outputs": ["prunedint8_sweep_t2v_608x352_10s_00001_.mp4"], "error": null, "ts": "2026-08-06T04:27:19Z"} diff --git a/docs/RESEARCH.md b/docs/RESEARCH.md new file mode 100644 index 0000000..412763e --- /dev/null +++ b/docs/RESEARCH.md @@ -0,0 +1,163 @@ +# MiniMax H3 — research findings + +Compiled 2026-08-06. Sources linked inline; benchmark numbers we measured +ourselves live in `bench/results.csv` and `docs/BENCHMARKS.md`. + +## The model + +[MiniMax H3](https://www.minimax.io/blog/minimax-h3) is an open-weights +omni-modal video model: one 33B dense "Omni-Transformer" DiT (50 layers, hidden +5376) jointly denoises video + **native stereo audio** latents in a single +forward pass, conditioned on hidden states from a **Qwen3-VL-32B** text encoder +(layer 50). CFG-distilled — no negative prompt / CFG pass. Output is 24 fps +fixed, 4–15 s, native canvas ~1 MP (2K comes from a separate regenerate/upscale +module in the hosted API). Video VAE: 16x spatial / 4x temporal. Audio VAE: +stereo 32 kHz, 40 latent fps. + +Two task checkpoints, each shipped in several precisions: + +- **fl2va** — text-to-video and first/last-frame conditioning (covers t2v + i2v) +- **ref2va** — reference-to-video: up to 9 ref images, 3 ref videos (with + soundtracks), 3 standalone audio clips, woven in via `` / + `