mirror of
https://github.com/storytold/cloud-worker.git
synced 2026-10-09 00:09:43 +00:00
MiniMax H3 on RunPod: install/download scripts, ComfyUI wiring, benchmark harness, research docs
- scripts/pod: weight downloads (workspace + tmpfs overflow for volume quota), extra_model_paths.yaml, headless ComfyUI launcher, test image generator, one-shot pod installer - scripts/local: auto-reconnecting port forward for the ComfyUI panel - bench: API-based harness for t2v/i2v/ref2v across durations, resolutions, and weight families; CSV/JSONL results - docs: WEIGHTS.md (what's downloaded where), RESEARCH.md (ComfyUI guides, no-Comfy options via SGLang/vLLM/diffusers, concurrency model, serverless) Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -1 +1,44 @@
|
||||
cloud-worker
|
||||
# cloud-worker
|
||||
|
||||
Tooling for running the **MiniMax H3** open-weights video model (native stereo
|
||||
audio, 24 fps, 5–15 s, ~1 MP native) on RunPod GPU machines via ComfyUI.
|
||||
|
||||
Current test rig: 1x B200 (180 GB) pod — connection details in `pod-info.txt`.
|
||||
|
||||
## Layout
|
||||
|
||||
- `scripts/pod/` — runs on the pod
|
||||
- `install-minimax-h3.sh` — one-shot setup on a fresh pod (after `install-comfy.sh`)
|
||||
- `download-weights.sh` — all weight families → `/workspace/minimax-h3/models`
|
||||
- `download-weights-tmpfs.sh` — overflow files → `/dev/shm` (network-volume
|
||||
quota workaround; ephemeral, re-run after pod restart)
|
||||
- `extra_model_paths.yaml` — installed to `/ComfyUI/extra_model_paths.yaml`
|
||||
- `start-comfyui.sh` — headless ComfyUI in tmux on 127.0.0.1:8188
|
||||
- `make-test-images.py` — synthetic reference/first-frame images → `/ComfyUI/input`
|
||||
- `scripts/local/port-forward.sh` — persistent tunnel `localhost:8188` → pod
|
||||
ComfyUI panel (auto-reconnects)
|
||||
- `bench/minimax_bench.py` — API-based benchmark harness (t2v / i2v / ref2v ×
|
||||
duration × resolution × weight family); appends to `bench/results.csv` +
|
||||
`bench/results.jsonl`
|
||||
- `docs/WEIGHTS.md` — which weights are downloaded, where, and which family
|
||||
- `docs/RESEARCH.md` — model/ComfyUI findings, running without ComfyUI
|
||||
(SGLang / vLLM-Omni / diffusers), concurrency model, RunPod serverless
|
||||
- `docs/BENCHMARKS.md` — measured results on the B200
|
||||
|
||||
## Quick start
|
||||
|
||||
```bash
|
||||
# tunnel to the ComfyUI panel (leave running)
|
||||
scripts/local/port-forward.sh &
|
||||
open http://localhost:8188
|
||||
|
||||
# one-off generation through the API
|
||||
python3 bench/minimax_bench.py --task t2v --width 864 --height 480 --seconds 5
|
||||
|
||||
# smoke suite / full sweep
|
||||
python3 bench/minimax_bench.py --suite quick
|
||||
python3 bench/minimax_bench.py --suite sweep --prefix prunedint8_
|
||||
```
|
||||
|
||||
On the pod, ComfyUI runs in tmux session `comfyui`
|
||||
(`tmux attach -t comfyui`), logs at `/workspace/minimax-h3/comfyui.log`.
|
||||
|
||||
@@ -0,0 +1,334 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Benchmark MiniMax H3 video generation through the ComfyUI API.
|
||||
|
||||
Builds API-format graphs programmatically (no workflow JSON needed) for the
|
||||
three local H3 task types and times each generation:
|
||||
|
||||
t2v MiniMaxH3ImageToVideo with no frames connected (pure text)
|
||||
i2v MiniMaxH3ImageToVideo with a first_frame image
|
||||
ref2v MiniMaxH3ReferenceToVideo with one reference image
|
||||
|
||||
Model facts baked in (see docs/RESEARCH.md):
|
||||
- frame length must satisfy n % 17 == 5; trained range 124-362 (~5-15 s @ 24fps)
|
||||
- canvas: multiples of 32, native area <= 768*1344 (~1 MP)
|
||||
- reference sampler config: res_multistep + simple scheduler, 20 steps, no CFG
|
||||
|
||||
Usage examples (through the SSH tunnel, or on the pod with --host):
|
||||
python bench/minimax_bench.py --suite quick
|
||||
python bench/minimax_bench.py --suite sweep --model fl2va=minimax_h3_fl2va_pruned_int8_convrot.safetensors \
|
||||
--model ref2va=minimax_h3_ref2va_pruned_int8_convrot.safetensors
|
||||
python bench/minimax_bench.py --task t2v --width 864 --height 480 --seconds 5 --steps 20
|
||||
|
||||
Results are appended to bench/results.csv (one row per run) and full metadata
|
||||
to bench/results.jsonl. Output videos are saved on the ComfyUI side under
|
||||
output/bench/.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import json
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
DEFAULT_HOST = "http://127.0.0.1:8188"
|
||||
|
||||
TEXT_ENCODER = "qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors"
|
||||
VIDEO_VAE = "minimax_h3_video_vae_fp16.safetensors"
|
||||
AUDIO_VAE = "minimax_h3_audio_vae_fp32.safetensors"
|
||||
FL2VA_DEFAULT = "minimax_h3_fl2va_pruned_int8_convrot.safetensors"
|
||||
REF2VA_DEFAULT = "minimax_h3_ref2va_pruned_int8_convrot.safetensors"
|
||||
BENCH_IMAGE = "bench_ref_1344x768.png" # created by scripts/pod/make-test-images.py
|
||||
|
||||
T2V_PROMPT = (
|
||||
"Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, "
|
||||
"waves crashing on dark rocks below, seagulls circling, warm sunlight flares, "
|
||||
"sound of surf and wind, gentle orchestral swell."
|
||||
)
|
||||
I2V_PROMPT = (
|
||||
"The camera slowly pushes into the city skyline as dusk settles, window lights "
|
||||
"flickering on one by one, light wind, distant traffic hum and a soft synth pad."
|
||||
)
|
||||
REF2V_PROMPT = (
|
||||
"Use <Picture 1> as the setting. A slow cinematic pan across the sunset city "
|
||||
"skyline, clouds drifting, lights turning on in the towers, ambient city sounds."
|
||||
)
|
||||
|
||||
|
||||
def snap_length(seconds: float) -> int:
|
||||
"""Duration in seconds -> valid H3 frame count (24 fps, n % 17 == 5, snapped up)."""
|
||||
n = max(5, round(seconds * 24))
|
||||
return n + (5 - (n % 17)) % 17
|
||||
|
||||
|
||||
def build_graph(task, model_file, prompt, width, height, length, steps, seed,
|
||||
sampler="res_multistep", scheduler="simple", image=BENCH_IMAGE,
|
||||
ref_image_size="match", filename_prefix="bench/run"):
|
||||
g = {
|
||||
"1": {"class_type": "UNETLoader",
|
||||
"inputs": {"unet_name": model_file, "weight_dtype": "default"}},
|
||||
"2": {"class_type": "CLIPLoader",
|
||||
"inputs": {"clip_name": TEXT_ENCODER, "type": "minimax", "device": "default"}},
|
||||
"3": {"class_type": "VAELoader", "inputs": {"vae_name": VIDEO_VAE}},
|
||||
"4": {"class_type": "VAELoader", "inputs": {"vae_name": AUDIO_VAE}},
|
||||
"6": {"class_type": "KSamplerSelect", "inputs": {"sampler_name": sampler}},
|
||||
"7": {"class_type": "BasicScheduler",
|
||||
"inputs": {"model": ["1", 0], "scheduler": scheduler, "steps": steps,
|
||||
"denoise": 1.0}},
|
||||
"8": {"class_type": "RandomNoise", "inputs": {"noise_seed": seed}},
|
||||
"9": {"class_type": "BasicGuider",
|
||||
"inputs": {"model": ["1", 0], "conditioning": ["5", 0]}},
|
||||
"11": {"class_type": "SamplerCustomAdvanced",
|
||||
"inputs": {"noise": ["8", 0], "guider": ["9", 0], "sampler": ["6", 0],
|
||||
"sigmas": ["7", 0], "latent_image": ["5", 1]}},
|
||||
"12": {"class_type": "VAEDecode", "inputs": {"samples": ["11", 0], "vae": ["3", 0]}},
|
||||
"13": {"class_type": "VAEDecodeAudio", "inputs": {"samples": ["11", 0], "vae": ["4", 0]}},
|
||||
"14": {"class_type": "CreateVideo",
|
||||
"inputs": {"images": ["12", 0], "audio": ["13", 0], "fps": 24, "bit_depth": 8}},
|
||||
"15": {"class_type": "SaveVideo",
|
||||
"inputs": {"video": ["14", 0], "filename_prefix": filename_prefix,
|
||||
"format": "auto", "codec": "auto"}},
|
||||
}
|
||||
common = {"prompt": prompt, "width": width, "height": height, "length": length}
|
||||
if task == "t2v":
|
||||
g["5"] = {"class_type": "MiniMaxH3ImageToVideo",
|
||||
"inputs": {"clip": ["2", 0], "vae": ["3", 0], **common}}
|
||||
elif task == "i2v":
|
||||
g["10"] = {"class_type": "LoadImage", "inputs": {"image": image}}
|
||||
g["5"] = {"class_type": "MiniMaxH3ImageToVideo",
|
||||
"inputs": {"clip": ["2", 0], "vae": ["3", 0], "first_frame": ["10", 0],
|
||||
**common}}
|
||||
elif task == "ref2v":
|
||||
g["10"] = {"class_type": "LoadImage", "inputs": {"image": image}}
|
||||
g["5"] = {"class_type": "MiniMaxH3ReferenceToVideo",
|
||||
"inputs": {"clip": ["2", 0], "vae": ["3", 0], "audio_vae": ["4", 0],
|
||||
"ref_image_size": ref_image_size,
|
||||
"ref_images.ref_image_0": ["10", 0], **common}}
|
||||
else:
|
||||
raise ValueError(f"unknown task {task}")
|
||||
return g
|
||||
|
||||
|
||||
class Api:
|
||||
def __init__(self, host):
|
||||
self.host = host.rstrip("/")
|
||||
self.client_id = str(uuid.uuid4())
|
||||
|
||||
def _req(self, path, data=None, timeout=30):
|
||||
url = self.host + path
|
||||
body = json.dumps(data).encode() if data is not None else None
|
||||
req = urllib.request.Request(url, data=body,
|
||||
headers={"Content-Type": "application/json"} if body else {})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
return json.loads(r.read() or "{}")
|
||||
|
||||
def queue_prompt(self, graph):
|
||||
return self._req("/prompt", {"prompt": graph, "client_id": self.client_id})
|
||||
|
||||
def history(self, prompt_id):
|
||||
return self._req(f"/history/{prompt_id}")
|
||||
|
||||
def vram_used(self):
|
||||
try:
|
||||
s = self._req("/system_stats", timeout=5)
|
||||
d = s["devices"][0]
|
||||
return d["vram_total"] - d["vram_free"]
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def run_one(api, cfg, poll=2.0, timeout=3600):
|
||||
"""Submit one benchmark run; returns result dict."""
|
||||
graph = build_graph(**{k: v for k, v in cfg.items() if k in (
|
||||
"task", "model_file", "prompt", "width", "height", "length", "steps", "seed",
|
||||
"sampler", "scheduler", "image", "ref_image_size", "filename_prefix")})
|
||||
peak = {"vram": 0}
|
||||
stop = threading.Event()
|
||||
|
||||
def sample_vram():
|
||||
while not stop.is_set():
|
||||
v = api.vram_used()
|
||||
if v:
|
||||
peak["vram"] = max(peak["vram"], v)
|
||||
stop.wait(3)
|
||||
|
||||
t = threading.Thread(target=sample_vram, daemon=True)
|
||||
t.start()
|
||||
t0 = time.time()
|
||||
try:
|
||||
resp = api.queue_prompt(graph)
|
||||
except urllib.error.HTTPError as e:
|
||||
stop.set()
|
||||
detail = e.read().decode()[:2000]
|
||||
return {**cfg, "ok": False, "error": f"HTTP {e.code}: {detail}"}
|
||||
prompt_id = resp["prompt_id"]
|
||||
exec_start = exec_end = None
|
||||
status_str = None
|
||||
while time.time() - t0 < timeout:
|
||||
time.sleep(poll)
|
||||
h = api.history(prompt_id)
|
||||
if prompt_id in h:
|
||||
entry = h[prompt_id]
|
||||
status_str = entry.get("status", {}).get("status_str")
|
||||
for msg in entry.get("status", {}).get("messages", []):
|
||||
if msg[0] == "execution_start":
|
||||
exec_start = msg[1].get("timestamp")
|
||||
if msg[0] in ("execution_success", "execution_error"):
|
||||
exec_end = msg[1].get("timestamp")
|
||||
if status_str in ("success", "error"):
|
||||
break
|
||||
stop.set()
|
||||
wall = time.time() - t0
|
||||
exec_s = (exec_end - exec_start) / 1000 if exec_start and exec_end else None
|
||||
outputs = []
|
||||
if status_str == "success":
|
||||
for node_out in h[prompt_id].get("outputs", {}).values():
|
||||
for key in ("images", "video", "videos"):
|
||||
for f in node_out.get(key, []):
|
||||
outputs.append(f.get("filename"))
|
||||
return {**cfg, "ok": status_str == "success", "status": status_str,
|
||||
"wall_s": round(wall, 1),
|
||||
"exec_s": round(exec_s, 1) if exec_s else None,
|
||||
"peak_vram_gb": round(peak["vram"] / 1e9, 1) if peak["vram"] else None,
|
||||
"outputs": outputs,
|
||||
"error": None if status_str == "success" else json.dumps(
|
||||
h.get(prompt_id, {}).get("status", {}))[:2000]}
|
||||
|
||||
|
||||
def append_results(row, outdir):
|
||||
outdir.mkdir(parents=True, exist_ok=True)
|
||||
jl = outdir / "results.jsonl"
|
||||
with jl.open("a") as f:
|
||||
f.write(json.dumps({**row, "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())}) + "\n")
|
||||
csv_path = outdir / "results.csv"
|
||||
fields = ["ts", "label", "task", "model_file", "width", "height", "seconds", "length",
|
||||
"steps", "sampler", "scheduler", "ref_image_size", "warm", "ok", "wall_s",
|
||||
"exec_s", "peak_vram_gb", "outputs", "error"]
|
||||
new = not csv_path.exists()
|
||||
with csv_path.open("a", newline="") as f:
|
||||
w = csv.DictWriter(f, fieldnames=fields, extrasaction="ignore")
|
||||
if new:
|
||||
w.writeheader()
|
||||
w.writerow({**row, "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
||||
"outputs": ";".join(row.get("outputs") or [])})
|
||||
|
||||
|
||||
def default_prompt(task):
|
||||
return {"t2v": T2V_PROMPT, "i2v": I2V_PROMPT, "ref2v": REF2V_PROMPT}[task]
|
||||
|
||||
|
||||
def make_cfg(task, model_file, width, height, seconds, steps, seed, label, warm,
|
||||
sampler="res_multistep", scheduler="simple", ref_image_size="match"):
|
||||
return {
|
||||
"label": label, "task": task, "model_file": model_file,
|
||||
"prompt": default_prompt(task), "width": width, "height": height,
|
||||
"seconds": seconds, "length": snap_length(seconds), "steps": steps,
|
||||
"seed": seed, "sampler": sampler, "scheduler": scheduler,
|
||||
"image": BENCH_IMAGE, "ref_image_size": ref_image_size, "warm": warm,
|
||||
"filename_prefix": f"bench/{label}",
|
||||
}
|
||||
|
||||
|
||||
def suite_quick(fl2va, ref2va, prefix=""):
|
||||
"""Small smoke suite: one tiny run per task type."""
|
||||
return [
|
||||
make_cfg("t2v", fl2va, 864, 480, 5, 20, 1, f"{prefix}quick_t2v", warm=False),
|
||||
make_cfg("i2v", fl2va, 864, 480, 5, 20, 1, f"{prefix}quick_i2v", warm=True),
|
||||
make_cfg("ref2v", ref2va, 864, 480, 5, 20, 1, f"{prefix}quick_ref2v", warm=False),
|
||||
]
|
||||
|
||||
|
||||
def suite_sweep(fl2va, ref2va, prefix=""):
|
||||
"""Full sweep: tasks x durations x resolutions at fixed 20 steps.
|
||||
|
||||
Resolution ladder (16:9, multiples of 32, up to the 1MP native cap):
|
||||
608x352 (0.2MP draft), 864x480 (0.4MP default), 1344x768 (1MP max)
|
||||
Durations: 5 s (124 fr), 10 s (243 fr), 15 s (362 fr) - the trained range.
|
||||
"""
|
||||
resolutions = [(608, 352), (864, 480), (1344, 768)]
|
||||
durations = [5, 10, 15]
|
||||
cfgs = []
|
||||
warm_fl = False
|
||||
for w, h in resolutions:
|
||||
for sec in durations:
|
||||
cfgs.append(make_cfg("t2v", fl2va, w, h, sec, 20, 1,
|
||||
f"{prefix}sweep_t2v_{w}x{h}_{sec}s", warm=warm_fl))
|
||||
warm_fl = True
|
||||
# i2v: model already warm; vary duration at default res, plus max res spot check
|
||||
for sec in durations:
|
||||
cfgs.append(make_cfg("i2v", fl2va, 864, 480, sec, 20, 1,
|
||||
f"{prefix}sweep_i2v_864x480_{sec}s", warm=True))
|
||||
cfgs.append(make_cfg("i2v", fl2va, 1344, 768, 5, 20, 1,
|
||||
f"{prefix}sweep_i2v_1344x768_5s", warm=True))
|
||||
# ref2v: separate weights -> first run is a model swap (cold)
|
||||
warm_ref = False
|
||||
for sec in durations:
|
||||
cfgs.append(make_cfg("ref2v", ref2va, 864, 480, sec, 20, 1,
|
||||
f"{prefix}sweep_ref2v_864x480_{sec}s", warm=warm_ref))
|
||||
warm_ref = True
|
||||
cfgs.append(make_cfg("ref2v", ref2va, 1344, 768, 5, 20, 1,
|
||||
f"{prefix}sweep_ref2v_1344x768_5s", warm=True))
|
||||
return cfgs
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--host", default=DEFAULT_HOST)
|
||||
ap.add_argument("--suite", choices=["quick", "sweep"])
|
||||
ap.add_argument("--task", choices=["t2v", "i2v", "ref2v"])
|
||||
ap.add_argument("--model", action="append", default=[],
|
||||
help="override model files, e.g. fl2va=file.safetensors (repeatable)")
|
||||
ap.add_argument("--width", type=int, default=864)
|
||||
ap.add_argument("--height", type=int, default=480)
|
||||
ap.add_argument("--seconds", type=float, default=5)
|
||||
ap.add_argument("--steps", type=int, default=20)
|
||||
ap.add_argument("--seed", type=int, default=1)
|
||||
ap.add_argument("--sampler", default="res_multistep")
|
||||
ap.add_argument("--scheduler", default="simple")
|
||||
ap.add_argument("--ref-image-size", default="match", choices=["match", "max"])
|
||||
ap.add_argument("--label", default=None)
|
||||
ap.add_argument("--prefix", default="", help="label prefix for suites, e.g. prunedint8_")
|
||||
ap.add_argument("--warm", action="store_true", help="mark run as warm (model preloaded)")
|
||||
ap.add_argument("--outdir", default=str(Path(__file__).parent))
|
||||
args = ap.parse_args()
|
||||
|
||||
models = {"fl2va": FL2VA_DEFAULT, "ref2va": REF2VA_DEFAULT}
|
||||
for m in args.model:
|
||||
k, v = m.split("=", 1)
|
||||
models[k] = v
|
||||
|
||||
api = Api(args.host)
|
||||
if args.suite == "quick":
|
||||
cfgs = suite_quick(models["fl2va"], models["ref2va"], args.prefix)
|
||||
elif args.suite == "sweep":
|
||||
cfgs = suite_sweep(models["fl2va"], models["ref2va"], args.prefix)
|
||||
elif args.task:
|
||||
model = models["ref2va"] if args.task == "ref2v" else models["fl2va"]
|
||||
label = args.label or f"{args.task}_{args.width}x{args.height}_{args.seconds}s"
|
||||
cfgs = [make_cfg(args.task, model, args.width, args.height, args.seconds,
|
||||
args.steps, args.seed, label, args.warm,
|
||||
args.sampler, args.scheduler, args.ref_image_size)]
|
||||
else:
|
||||
ap.error("need --suite or --task")
|
||||
|
||||
outdir = Path(args.outdir)
|
||||
for i, cfg in enumerate(cfgs):
|
||||
print(f"[{i+1}/{len(cfgs)}] {cfg['label']}: {cfg['task']} {cfg['model_file']} "
|
||||
f"{cfg['width']}x{cfg['height']} {cfg['seconds']}s ({cfg['length']}fr) "
|
||||
f"{cfg['steps']} steps ...", flush=True)
|
||||
row = run_one(api, cfg)
|
||||
append_results(row, outdir)
|
||||
if row["ok"]:
|
||||
print(f" OK wall={row['wall_s']}s exec={row['exec_s']}s "
|
||||
f"peak_vram={row['peak_vram_gb']}GB -> {row['outputs']}", flush=True)
|
||||
else:
|
||||
print(f" FAILED: {str(row.get('error'))[:400]}", flush=True)
|
||||
print("done.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,6 @@
|
||||
ts,label,task,model_file,width,height,seconds,length,steps,sampler,scheduler,ref_image_size,warm,ok,wall_s,exec_s,peak_vram_gb,outputs,error
|
||||
2026-08-06T04:19:55Z,smoke_t2v,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,False,True,80.8,80.1,42.0,smoke_t2v_00001_.mp4,
|
||||
2026-08-06T04:20:48Z,smoke_i2v,i2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,True,True,37.9,36.3,45.0,smoke_i2v_00001_.mp4,
|
||||
2026-08-06T04:22:41Z,smoke_ref2v,ref2v,minimax_h3_ref2va_pruned_int8_convrot.safetensors,608,352,5.0,124,20,res_multistep,simple,match,False,True,113.3,111.4,67.0,smoke_ref2v_00001_.mp4,
|
||||
2026-08-06T04:26:04Z,prunedint8_sweep_t2v_608x352_5s,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,5,124,20,res_multistep,simple,match,False,True,10.9,9.6,64.4,prunedint8_sweep_t2v_608x352_5s_00001_.mp4,
|
||||
2026-08-06T04:27:19Z,prunedint8_sweep_t2v_608x352_10s,t2v,minimax_h3_fl2va_pruned_int8_convrot.safetensors,608,352,10,243,20,res_multistep,simple,match,True,True,75.9,74.6,68.4,prunedint8_sweep_t2v_608x352_10s_00001_.mp4,
|
||||
|
@@ -0,0 +1,5 @@
|
||||
{"label": "smoke_t2v", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/smoke_t2v", "ok": true, "status": "success", "wall_s": 80.8, "exec_s": 80.1, "peak_vram_gb": 42.0, "outputs": ["smoke_t2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:19:55Z"}
|
||||
{"label": "smoke_i2v", "task": "i2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "The camera slowly pushes into the city skyline as dusk settles, window lights flickering on one by one, light wind, distant traffic hum and a soft synth pad.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": true, "filename_prefix": "bench/smoke_i2v", "ok": true, "status": "success", "wall_s": 37.9, "exec_s": 36.3, "peak_vram_gb": 45.0, "outputs": ["smoke_i2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:20:48Z"}
|
||||
{"label": "smoke_ref2v", "task": "ref2v", "model_file": "minimax_h3_ref2va_pruned_int8_convrot.safetensors", "prompt": "Use <Picture 1> as the setting. A slow cinematic pan across the sunset city skyline, clouds drifting, lights turning on in the towers, ambient city sounds.", "width": 608, "height": 352, "seconds": 5.0, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/smoke_ref2v", "ok": true, "status": "success", "wall_s": 113.3, "exec_s": 111.4, "peak_vram_gb": 67.0, "outputs": ["smoke_ref2v_00001_.mp4"], "error": null, "ts": "2026-08-06T04:22:41Z"}
|
||||
{"label": "prunedint8_sweep_t2v_608x352_5s", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 5, "length": 124, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": false, "filename_prefix": "bench/prunedint8_sweep_t2v_608x352_5s", "ok": true, "status": "success", "wall_s": 10.9, "exec_s": 9.6, "peak_vram_gb": 64.4, "outputs": ["prunedint8_sweep_t2v_608x352_5s_00001_.mp4"], "error": null, "ts": "2026-08-06T04:26:04Z"}
|
||||
{"label": "prunedint8_sweep_t2v_608x352_10s", "task": "t2v", "model_file": "minimax_h3_fl2va_pruned_int8_convrot.safetensors", "prompt": "Cinematic aerial shot slowly orbiting a coastal lighthouse at golden hour, waves crashing on dark rocks below, seagulls circling, warm sunlight flares, sound of surf and wind, gentle orchestral swell.", "width": 608, "height": 352, "seconds": 10, "length": 243, "steps": 20, "seed": 1, "sampler": "res_multistep", "scheduler": "simple", "image": "bench_ref_1344x768.png", "ref_image_size": "match", "warm": true, "filename_prefix": "bench/prunedint8_sweep_t2v_608x352_10s", "ok": true, "status": "success", "wall_s": 75.9, "exec_s": 74.6, "peak_vram_gb": 68.4, "outputs": ["prunedint8_sweep_t2v_608x352_10s_00001_.mp4"], "error": null, "ts": "2026-08-06T04:27:19Z"}
|
||||
@@ -0,0 +1,163 @@
|
||||
# MiniMax H3 — research findings
|
||||
|
||||
Compiled 2026-08-06. Sources linked inline; benchmark numbers we measured
|
||||
ourselves live in `bench/results.csv` and `docs/BENCHMARKS.md`.
|
||||
|
||||
## The model
|
||||
|
||||
[MiniMax H3](https://www.minimax.io/blog/minimax-h3) is an open-weights
|
||||
omni-modal video model: one 33B dense "Omni-Transformer" DiT (50 layers, hidden
|
||||
5376) jointly denoises video + **native stereo audio** latents in a single
|
||||
forward pass, conditioned on hidden states from a **Qwen3-VL-32B** text encoder
|
||||
(layer 50). CFG-distilled — no negative prompt / CFG pass. Output is 24 fps
|
||||
fixed, 4–15 s, native canvas ~1 MP (2K comes from a separate regenerate/upscale
|
||||
module in the hosted API). Video VAE: 16x spatial / 4x temporal. Audio VAE:
|
||||
stereo 32 kHz, 40 latent fps.
|
||||
|
||||
Two task checkpoints, each shipped in several precisions:
|
||||
|
||||
- **fl2va** — text-to-video and first/last-frame conditioning (covers t2v + i2v)
|
||||
- **ref2va** — reference-to-video: up to 9 ref images, 3 ref videos (with
|
||||
soundtracks), 3 standalone audio clips, woven in via `<Picture i>` /
|
||||
`<Video k>` / `<Audio j>` prompt tags
|
||||
|
||||
**License caveat**: the MiniMax H3 Community License restricts open-weight use
|
||||
to the **EU, UK, South Korea, and US**; elsewhere requires a license from
|
||||
MiniMax ([Q&A](https://huggingface.co/MiniMaxAI/MiniMax-H3/blob/main/docs/QA-about-License.md)).
|
||||
|
||||
## Weight families ("bf16 / int8 / int8 pruned")
|
||||
|
||||
- **bf16** — full precision, 66 GB per model line. Reference quality.
|
||||
- **int8_convrot** — 34 GB. [ConvRot](https://arxiv.org/pdf/2512.03673) is
|
||||
rotation-based post-training quantization: a group-wise Hadamard rotation
|
||||
(expressed as a conv, hence the name) redistributes activation outliers so
|
||||
int8 quantizes near-losslessly. Community consensus: "near-lossless".
|
||||
- **pruned_int8_convrot** — 21 GB. "Pruned" here is *not* layer dropping: the
|
||||
DiT's adaLN modulation weights (~40% of params) are replaced by a
|
||||
functionally-equivalent lookup table. Comfy/MiniMax claim no quality loss;
|
||||
HF discussions call the pruned line "a bit more experimental".
|
||||
- (Also upstream: pruned_bf16 40 GB, pruned_fp8_scaled 21 GB, and community
|
||||
INT4/NVFP4/W4A8 quants — see [awesome-minimax-H3](https://github.com/wildminder/awesome-minimax-H3).)
|
||||
|
||||
Full stack VRAM classes (community): bf16 → 80 GB+; int8 → 48 GB; pruned int8 →
|
||||
fits 32 GB cards; INT4-class community quants reach 12–16 GB.
|
||||
|
||||
## ComfyUI integration
|
||||
|
||||
- Support merged day-0: [PR #15224](https://github.com/Comfy-Org/ComfyUI/pull/15224);
|
||||
int8_convrot VAE decode (~1.5x faster) in [PR #15334](https://github.com/Comfy-Org/ComfyUI/pull/15334).
|
||||
- Official tutorial: <https://docs.comfy.org/tutorials/video/minimax/minimax-h3>;
|
||||
blog: <https://blog.comfy.org/p/minimax-h3-day-0-support-in-comfyui>;
|
||||
community wiki: <https://comfyui-wiki.com/en/tutorial/advanced/video/minimax/minimax-h3>.
|
||||
- Nodes (`comfy_extras/nodes_minimax_h3.py`):
|
||||
- `MiniMaxH3ImageToVideo` — **both t2v and i2v**: optional `first_frame` /
|
||||
`last_frame`; no frames connected = pure t2v. Outputs conditioning + packed
|
||||
AV latent.
|
||||
- `MiniMaxH3ReferenceToVideo` — ref2v; `ref_image_size` `match` (fast) vs
|
||||
`max` (2048px refs, several times slower — ref tokens ride every step).
|
||||
- `EmptyMiniMaxH3LatentAV`, `MiniMaxH3SigmaShift` (defaults video 12.0 /
|
||||
audio 3.0 — already model defaults; node exists for overrides).
|
||||
- Template sampling config: `res_multistep` sampler + `simple` scheduler,
|
||||
**20 steps**, denoise 1.0, `BasicGuider` (no CFG), `CreateVideo` fps=24.
|
||||
Wiki: quality degrades <15 steps, marginal gains >25; `beta`/`normal`
|
||||
scheduler reportedly beats `simple` for reference-heavy r2v prompts.
|
||||
- **Frame length rule**: `n % 17 == 5`, i.e. 5, 22, …, 124 (≈5 s), 243 (≈10 s),
|
||||
362 (≈15 s). Trained range 124–362; longer untested. Templates compute
|
||||
`max(5, round(sec*24))` snapped **up** to the grid.
|
||||
- **Resolution rule**: multiples of 32 per axis, native area ≤ 768×1344 ≈ 1 MP.
|
||||
Practical floor ~384p. fps is fixed at 24 (baked into the model).
|
||||
- Known pitfalls (Aug 2026): global `--use-sage-attention` produces noise on H3
|
||||
([#15263](https://github.com/Comfy-Org/ComfyUI/issues/15263)) — use the KJNodes
|
||||
"Patch Sage Attention" node instead (~2x speedup reported); EasyCache degrades
|
||||
the audio stream ([#15326](https://github.com/Comfy-Org/ComfyUI/issues/15326));
|
||||
tiled-VAE and ref-video-encode OOM issues (#15312/#15274/#15246).
|
||||
|
||||
## Concurrency: what actually maximizes throughput
|
||||
|
||||
The finding that shapes everything: **one H3 generation saturates the GPU**
|
||||
(DiT denoise ≈ 88% of request wall time per vLLM's own profiling). Consequences:
|
||||
|
||||
- Running N generations concurrently on one GPU ≈ N× the latency of one — no
|
||||
throughput win (SGLang's 2-outputs-per-prompt numbers scale near-linearly).
|
||||
- Neither SGLang nor vLLM-Omni does cross-request batching for H3
|
||||
("one generation per diffusion batch"), and ComfyUI's H3 latent node doesn't
|
||||
expose batch_size.
|
||||
- ComfyUI executes its queue **serially** — one workflow at a time per instance.
|
||||
|
||||
So "as many concurrent generations as possible" = **one resident worker per
|
||||
GPU, horizontal scale-out**, plus multi-GPU parallelism to cut per-video
|
||||
latency where it matters. On a multi-GPU pod you can run one ComfyUI instance
|
||||
per GPU with `--cuda-device N` pinning and separate ports.
|
||||
|
||||
## Running H3 without ComfyUI
|
||||
|
||||
Official repo [MiniMax-AI/MiniMax-H3](https://github.com/MiniMax-AI/MiniMax-H3)
|
||||
documents four supported paths (original weights:
|
||||
[MiniMaxAI/MiniMax-H3](https://huggingface.co/MiniMaxAI/MiniMax-H3)):
|
||||
|
||||
1. **SGLang — recommended resident server.**
|
||||
[Cookbook](https://docs.sglang.io/cookbook/diffusion/MiniMax/MiniMax-H3).
|
||||
`uv pip install "sglang[diffusion]" --prerelease=allow`, then e.g.
|
||||
`sglang serve --model-path MiniMaxAI/MiniMax-H3 --model-variant fl2va
|
||||
--num-gpus 4 --ulysses-degree 4 --performance-mode speed`.
|
||||
Model stays in VRAM; async REST job API (`POST /v1/videos` → poll →
|
||||
download). Published perf: 4x H200, 1344×768, 124 fr, 50 steps = **75 s**
|
||||
(54 s with Cache-DiT); 4x H100 (TP2+Ulysses2) is the fastest verified combo;
|
||||
online FP8 on B200/B300. Supports `num_outputs_per_prompt`.
|
||||
2. **vLLM-Omni** — [recipe](https://recipes.vllm.ai/MiniMaxAI/MiniMax-H3);
|
||||
sync endpoint `POST /v1/videos/sync`; ROCm docker for MI300X. FP8 not yet.
|
||||
3. **diffusers — merged Aug 5 2026** ([PR #14355](https://github.com/huggingface/diffusers/pull/14355)).
|
||||
Modular Diffusers only (no classic `DiffusionPipeline`):
|
||||
`ModularPipeline.from_pretrained("MiniMaxAI/MiniMax-H3")` →
|
||||
`MiniMaxH3ModularPipeline`; classes `MiniMaxH3Transformer3D`,
|
||||
`AutoencoderKLMiniMaxH3`, `MiniMaxH3Scheduler`. Install from git main until
|
||||
released. Best for custom logic; you manage offload/parallelism yourself.
|
||||
4. **ComfyUI API mode** — what this repo uses today: headless `main.py`,
|
||||
models resident between jobs, `POST /prompt` + `/history` (see
|
||||
`bench/minimax_bench.py` for a zero-dependency client). Serial queue.
|
||||
|
||||
Also: [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio)
|
||||
(supports H3 incl. NF4, low-VRAM Python path). No TensorRT/xDiT/FastVideo
|
||||
support found.
|
||||
|
||||
**Keep-loaded vs serverless**: keeping the model resident is what ComfyUI/SGLang
|
||||
already do. For scale-to-zero economics:
|
||||
|
||||
- **RunPod Serverless has B200** (180 GB, $0.00240/s ≈ $8.64/hr; H200 $0.00155/s,
|
||||
H100 $0.00116/s — [docs](https://docs.runpod.io/serverless/endpoints/endpoint-configurations)).
|
||||
- Weights on a network volume load at ~200–400 MB/s → **cold start ~3–8 min**
|
||||
for 40–115 GB of weights. FlashBoot only helps with steady traffic, not true
|
||||
scale-from-zero ([RunPod blog](https://www.runpod.io/blog/serverless-gpu-cold-starts-flashboot)).
|
||||
- Practical setup: [worker-comfyui](https://github.com/runpod-workers/worker-comfyui)
|
||||
or a custom `runpod` SDK handler wrapping SGLang/diffusers; **min-workers 1**
|
||||
turns it into a resident server with burst capacity — usually cheaper than
|
||||
eating multi-minute cold starts given 1–5 min generation times.
|
||||
- Managed alternative: [fal.ai hosts H3](https://fal.ai/minimax-h3).
|
||||
|
||||
## Published performance reference points
|
||||
|
||||
| Setup | Config | Time |
|
||||
|---|---|---|
|
||||
| 4x H100 80GB (SGLang TP2+Ulysses2) | 1344×768, 124 fr, 50 steps | 13.3 s |
|
||||
| 8x B300 (SGLang, bf16 / online FP8) | same | 19.0 / 18.0 s |
|
||||
| 4x H200 (SGLang Ulysses4) | same | 75.1 s (53.7 s Cache-DiT) |
|
||||
| 2x RTX 5090 (layerwise offload, ~380 GB host RAM) | same | 560 s |
|
||||
| RTX 4090 Laptop 16 GB (SageAttention) | 960×540, 5 s, 20 steps | 182 s |
|
||||
| RTX 3060 12 GB (heavy offload) | 864×480, 124 fr, 20 steps | <9 min |
|
||||
|
||||
Our single-B200 ComfyUI numbers: see `docs/BENCHMARKS.md`.
|
||||
|
||||
## Recommendations for storyteller
|
||||
|
||||
1. **Today (single B200 pod)**: ComfyUI headless + API is fine for dev and
|
||||
already keeps weights resident; use `bench/minimax_bench.py` patterns for
|
||||
programmatic generation. Pruned int8 quality is strong and it's the
|
||||
fastest/smallest; keep bf16 for quality A/Bs.
|
||||
2. **Production serving**: SGLang serve is the purpose-built path (async job
|
||||
API, multi-GPU, Cache-DiT). Evaluate 4x H100/H200 vs 1x B200 per-video
|
||||
economics — multi-GPU Ulysses cuts latency 4–6x for ~4x the GPU cost.
|
||||
3. **Concurrency**: scale horizontally (1 worker/GPU). Don't co-schedule two
|
||||
generations on one GPU.
|
||||
4. **Serverless**: viable on RunPod with B200 + network volume + min-workers 1;
|
||||
pure scale-from-zero pays 3–8 min cold starts. Consider baking pruned-int8
|
||||
weights (~42 GB stack) into the image to cut that.
|
||||
@@ -0,0 +1,57 @@
|
||||
# MiniMax H3 — weights log (B200 pod `c306014998a3`, 216.243.220.136)
|
||||
|
||||
All files from **[Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3)** (ComfyUI repack of
|
||||
[MiniMaxAI/MiniMax-H3](https://huggingface.co/MiniMaxAI/MiniMax-H3)). Two model lines:
|
||||
**fl2va** (text-to-video + first/last-frame i2v) and **ref2va** (reference-to-video).
|
||||
ComfyUI finds both storage locations via `/ComfyUI/extra_model_paths.yaml`
|
||||
(installed from `scripts/pod/extra_model_paths.yaml`).
|
||||
|
||||
## Storage locations
|
||||
|
||||
| Path | Medium | Notes |
|
||||
|---|---|---|
|
||||
| `/workspace/minimax-h3/models/` | RunPod network volume | **Persistent.** Volume has a ~100 GB quota — it is full; don't add files without removing others. |
|
||||
| `/dev/shm/minimax-h3-models/` | tmpfs (RAM, 176 GB) | **EPHEMERAL** — wiped on pod restart. Refill with `bash /workspace/minimax-h3/download-weights-tmpfs.sh`. |
|
||||
|
||||
## Downloaded files
|
||||
|
||||
### Persistent — `/workspace/minimax-h3/models/`
|
||||
|
||||
| File | Family | Size |
|
||||
|---|---|---|
|
||||
| `diffusion_models/minimax_h3_ref2va_pruned_int8_convrot.safetensors` | **int8 pruned** (ref2va) | 21 GB |
|
||||
| `diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors` | **int8 pruned** (fl2va) | 21 GB |
|
||||
| `diffusion_models/minimax_h3_ref2va_int8_convrot.safetensors` | **int8** (ref2va) | 34 GB |
|
||||
| `text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors` | text encoder (Qwen3-VL-32B, nvfp4 AWQ) | 16 GB |
|
||||
| `vae/minimax_h3_video_vae_fp16.safetensors` | video VAE fp16 | 5.2 GB |
|
||||
| `vae/minimax_h3_audio_vae_fp32.safetensors` | audio VAE fp32 | 0.6 GB |
|
||||
|
||||
### Ephemeral (tmpfs) — `/dev/shm/minimax-h3-models/`
|
||||
|
||||
| File | Family | Size |
|
||||
|---|---|---|
|
||||
| `diffusion_models/minimax_h3_fl2va_int8_convrot.safetensors` | **int8** (fl2va) | 34 GB |
|
||||
| `diffusion_models/minimax_h3_ref2va_bf16.safetensors` | **bf16** (ref2va) | 66 GB |
|
||||
| `diffusion_models/minimax_h3_fl2va_bf16.safetensors` | **bf16** (fl2va) | 66 GB |
|
||||
|
||||
The int8 family is split across both locations purely because of the volume
|
||||
quota: ref2va_int8 landed on the volume before it filled, fl2va_int8 overflowed
|
||||
to tmpfs.
|
||||
|
||||
## Not downloaded (exist upstream)
|
||||
|
||||
| File | Size | Why skipped |
|
||||
|---|---|---|
|
||||
| `diffusion_models/minimax_h3_{fl2va,ref2va}_pruned_bf16.safetensors` | 40 GB each | 4th family (pruned but unquantized); not in the 3 families requested |
|
||||
| `diffusion_models/minimax_h3_{fl2va,ref2va}_pruned_fp8_scaled.safetensors` | 21 GB each | 5th family; same size class as pruned int8 |
|
||||
| `text_encoders/qwen3vl_32b_minimax_h3_bf16.safetensors` | 52 GB | nvfp4_awq TE is the one every official template uses |
|
||||
| `text_encoders/qwen3vl_32b_minimax_h3_int8_convrot.safetensors` | 27 GB | same |
|
||||
|
||||
## History / gotchas
|
||||
|
||||
- 2026-08-06: original manual download of ref2va_pruned_int8 had a broken
|
||||
filename (`...safetensors?download=true` from a wget of the HF web URL); byte
|
||||
size matched HF exactly so it was renamed + moved to the volume, not re-downloaded.
|
||||
- 2026-08-06: bf16 + fl2va_int8 downloads to the volume failed with
|
||||
`Disk quota exceeded` at ~93 GB used → tmpfs overflow scheme added.
|
||||
- Downloads log: `/workspace/minimax-h3/download.log`.
|
||||
Executable
+15
@@ -0,0 +1,15 @@
|
||||
#!/bin/bash
|
||||
# Persistent port forward: localhost:8188 -> pod ComfyUI.
|
||||
# Auto-reconnects if the tunnel drops. Ctrl-C to stop.
|
||||
# Requires the echelon key loaded in your ssh agent.
|
||||
POD_HOST=216.243.220.136
|
||||
POD_PORT=14278
|
||||
while true; do
|
||||
ssh -N -L 8188:127.0.0.1:8188 \
|
||||
-o ServerAliveInterval=30 -o ServerAliveCountMax=3 \
|
||||
-o ExitOnForwardFailure=yes \
|
||||
-o StrictHostKeyChecking=accept-new \
|
||||
-p "$POD_PORT" root@"$POD_HOST"
|
||||
echo "tunnel dropped, reconnecting in 3s..."
|
||||
sleep 3
|
||||
done
|
||||
@@ -0,0 +1,41 @@
|
||||
#!/bin/bash
|
||||
# Overflow download for the weight files that don't fit under the /workspace
|
||||
# network-volume quota (~100GB on the current pod). Puts them on tmpfs
|
||||
# (/dev/shm, 176GB on the B200 pod) — EPHEMERAL: lost on pod restart, so this
|
||||
# is for benchmarking sessions. Re-run after any pod restart.
|
||||
#
|
||||
# tmux new -s downloads-shm -d 'bash /workspace/minimax-h3/download-weights-tmpfs.sh'
|
||||
#
|
||||
# ComfyUI sees this path via the minimax_shm section of extra_model_paths.yaml.
|
||||
set -uo pipefail
|
||||
|
||||
REPO="Comfy-Org/MiniMax-H3"
|
||||
DEST="/dev/shm/minimax-h3-models"
|
||||
LOG="/workspace/minimax-h3/download.log"
|
||||
|
||||
mkdir -p "$DEST/diffusion_models"
|
||||
export HF_HUB_ENABLE_HF_TRANSFER=1
|
||||
|
||||
log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; }
|
||||
|
||||
dl() {
|
||||
local file="$1"
|
||||
if [ -f "$DEST/$file" ]; then
|
||||
log "SKIP $file (already present in shm)"
|
||||
return 0
|
||||
fi
|
||||
log "START $file -> tmpfs"
|
||||
if hf download "$REPO" "$file" --local-dir "$DEST" >>"$LOG" 2>&1; then
|
||||
log "DONE $file ($(du -h "$DEST/$file" | cut -f1))"
|
||||
else
|
||||
log "FAILED $file"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
log "=== tmpfs overflow download starting ==="
|
||||
dl diffusion_models/minimax_h3_fl2va_int8_convrot.safetensors
|
||||
dl diffusion_models/minimax_h3_ref2va_bf16.safetensors
|
||||
dl diffusion_models/minimax_h3_fl2va_bf16.safetensors
|
||||
rm -rf "$DEST/.cache"
|
||||
log "=== tmpfs overflow downloads complete ==="
|
||||
Executable
+58
@@ -0,0 +1,58 @@
|
||||
#!/bin/bash
|
||||
# Download all MiniMax H3 weight families from Comfy-Org/MiniMax-H3 to the
|
||||
# RunPod network volume. Run on the pod, ideally inside tmux:
|
||||
# tmux new -s downloads -d 'bash /workspace/minimax-h3/download-weights.sh'
|
||||
#
|
||||
# Ordered so ComfyUI becomes usable as early as possible:
|
||||
# phase 1: VAEs + nvfp4 text encoder + pruned-int8 diffusion models (smallest family)
|
||||
# phase 2: int8_convrot family
|
||||
# phase 3: bf16 family (largest)
|
||||
#
|
||||
# Everything lands under /workspace/minimax-h3/models/{diffusion_models,text_encoders,vae}
|
||||
# which ComfyUI picks up via extra_model_paths.yaml.
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
REPO="Comfy-Org/MiniMax-H3"
|
||||
DEST="/workspace/minimax-h3/models"
|
||||
LOG="/workspace/minimax-h3/download.log"
|
||||
|
||||
mkdir -p "$DEST"
|
||||
export HF_HUB_ENABLE_HF_TRANSFER=1
|
||||
|
||||
log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; }
|
||||
|
||||
dl() {
|
||||
local file="$1"
|
||||
# hf download preserves the repo subpath (diffusion_models/..., vae/...) under --local-dir
|
||||
if [ -f "$DEST/$file" ]; then
|
||||
log "SKIP $file (already present)"
|
||||
return 0
|
||||
fi
|
||||
log "START $file"
|
||||
if hf download "$REPO" "$file" --local-dir "$DEST" >>"$LOG" 2>&1; then
|
||||
log "DONE $file ($(du -h "$DEST/$file" | cut -f1))"
|
||||
else
|
||||
log "FAILED $file"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
log "=== MiniMax H3 weight download run starting ==="
|
||||
|
||||
# --- Phase 1: minimum viable set (VAEs, text encoder, pruned int8 models) ---
|
||||
dl vae/minimax_h3_video_vae_fp16.safetensors
|
||||
dl vae/minimax_h3_audio_vae_fp32.safetensors
|
||||
dl text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors
|
||||
dl diffusion_models/minimax_h3_ref2va_pruned_int8_convrot.safetensors
|
||||
dl diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors
|
||||
|
||||
# --- Phase 2: int8_convrot (mid-size family) ---
|
||||
dl diffusion_models/minimax_h3_ref2va_int8_convrot.safetensors
|
||||
dl diffusion_models/minimax_h3_fl2va_int8_convrot.safetensors
|
||||
|
||||
# --- Phase 3: bf16 (full-precision family) ---
|
||||
dl diffusion_models/minimax_h3_ref2va_bf16.safetensors
|
||||
dl diffusion_models/minimax_h3_fl2va_bf16.safetensors
|
||||
|
||||
log "=== All downloads complete ==="
|
||||
@@ -0,0 +1,13 @@
|
||||
# ComfyUI extra model paths — points at the RunPod network volume where all
|
||||
# MiniMax H3 weights live. Install to /ComfyUI/extra_model_paths.yaml
|
||||
minimax_workspace:
|
||||
base_path: /workspace/minimax-h3/models
|
||||
diffusion_models: diffusion_models
|
||||
text_encoders: text_encoders
|
||||
vae: vae
|
||||
|
||||
# tmpfs overflow for files that exceed the network-volume quota (~100GB).
|
||||
# EPHEMERAL — refill with download-weights-tmpfs.sh after a pod restart.
|
||||
minimax_shm:
|
||||
base_path: /dev/shm/minimax-h3-models
|
||||
diffusion_models: diffusion_models
|
||||
@@ -0,0 +1,34 @@
|
||||
#!/bin/bash
|
||||
# One-shot MiniMax H3 setup for a fresh RunPod pod that already has ComfyUI at
|
||||
# /ComfyUI (see install-comfy.sh). Idempotent.
|
||||
#
|
||||
# Usage (from your machine):
|
||||
# scp -O -P <port> scripts/pod/{install-minimax-h3.sh,download-weights.sh,download-weights-tmpfs.sh,extra_model_paths.yaml,make-test-images.py} root@<pod>:/root/
|
||||
# ssh -p <port> root@<pod> 'bash /root/install-minimax-h3.sh'
|
||||
set -euo pipefail
|
||||
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
# deps
|
||||
pip install -q --break-system-packages hf_transfer 2>/dev/null || pip install -q hf_transfer
|
||||
|
||||
# layout + scripts on the persistent volume
|
||||
mkdir -p /workspace/minimax-h3/models
|
||||
cp "$HERE"/download-weights.sh "$HERE"/download-weights-tmpfs.sh \
|
||||
"$HERE"/make-test-images.py /workspace/minimax-h3/ 2>/dev/null || true
|
||||
cp "$HERE"/start-comfyui.sh /workspace/minimax-h3/ 2>/dev/null || true
|
||||
|
||||
# wire model paths into ComfyUI
|
||||
cp "$HERE"/extra_model_paths.yaml /ComfyUI/extra_model_paths.yaml
|
||||
|
||||
# test/reference images
|
||||
python3 /workspace/minimax-h3/make-test-images.py
|
||||
|
||||
# weights (phase 1 makes ComfyUI usable; later phases are the bigger families)
|
||||
tmux new -s downloads -d 'bash /workspace/minimax-h3/download-weights.sh' 2>/dev/null \
|
||||
|| echo "downloads session already exists"
|
||||
|
||||
# ComfyUI headless
|
||||
bash /workspace/minimax-h3/start-comfyui.sh
|
||||
|
||||
echo "Done. Watch downloads: tmux attach -t downloads ; tail -f /workspace/minimax-h3/download.log"
|
||||
@@ -0,0 +1,121 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate synthetic test/reference images into /ComfyUI/input.
|
||||
|
||||
Creates:
|
||||
- red_superboy_on_city_roof.png / mecha_dragon_lightning.png: stylized
|
||||
placeholders matching the refs in the installed r2v workflow so it runs
|
||||
out of the box.
|
||||
- bench_ref_1344x768.png: generic scene used by the benchmark harness for
|
||||
i2v (first frame) and ref2v runs.
|
||||
"""
|
||||
import math
|
||||
import random
|
||||
|
||||
from PIL import Image, ImageDraw, ImageFilter
|
||||
|
||||
OUT = "/ComfyUI/input"
|
||||
|
||||
|
||||
def city_backdrop(w, h, sky_top, sky_bottom, seed=7):
|
||||
rng = random.Random(seed)
|
||||
img = Image.new("RGB", (w, h))
|
||||
d = ImageDraw.Draw(img)
|
||||
for y in range(h):
|
||||
t = y / h
|
||||
d.line(
|
||||
[(0, y), (w, y)],
|
||||
fill=tuple(int(a + (b - a) * t) for a, b in zip(sky_top, sky_bottom)),
|
||||
)
|
||||
# skyline
|
||||
x = 0
|
||||
while x < w:
|
||||
bw = rng.randint(w // 20, w // 8)
|
||||
bh = rng.randint(h // 4, int(h * 0.55))
|
||||
d.rectangle([x, h - bh, x + bw, h], fill=(18, 16, 28))
|
||||
for wy in range(h - bh + 10, h - 10, 24):
|
||||
for wx in range(x + 8, x + bw - 8, 20):
|
||||
if rng.random() < 0.35:
|
||||
d.rectangle([wx, wy, wx + 8, wy + 12], fill=(255, 220, 120))
|
||||
x += bw + rng.randint(4, 24)
|
||||
return img
|
||||
|
||||
|
||||
def superboy():
|
||||
w, h = 1344, 768
|
||||
img = city_backdrop(w, h, (30, 20, 60), (120, 40, 80), seed=11)
|
||||
d = ImageDraw.Draw(img)
|
||||
# rooftop
|
||||
d.rectangle([0, int(h * 0.78), w, h], fill=(28, 26, 36))
|
||||
cx, cy = w // 2, int(h * 0.55)
|
||||
# cape
|
||||
d.polygon(
|
||||
[(cx - 28, cy - 40), (cx + 28, cy - 40), (cx + 95, cy + 130), (cx - 60, cy + 120)],
|
||||
fill=(190, 20, 30),
|
||||
)
|
||||
# body
|
||||
d.rectangle([cx - 26, cy - 35, cx + 26, cy + 70], fill=(40, 60, 160))
|
||||
# legs
|
||||
d.rectangle([cx - 24, cy + 70, cx - 6, cy + 170], fill=(190, 20, 30))
|
||||
d.rectangle([cx + 6, cy + 70, cx + 24, cy + 170], fill=(190, 20, 30))
|
||||
# arms on hips
|
||||
d.polygon([(cx - 26, cy - 25), (cx - 62, cy + 25), (cx - 26, cy + 32)], fill=(40, 60, 160))
|
||||
d.polygon([(cx + 26, cy - 25), (cx + 62, cy + 25), (cx + 26, cy + 32)], fill=(40, 60, 160))
|
||||
# head
|
||||
d.ellipse([cx - 24, cy - 88, cx + 24, cy - 40], fill=(250, 210, 170))
|
||||
d.arc([cx - 16, cy - 70, cx + 16, cy - 46], 20, 160, fill=(120, 60, 30), width=3)
|
||||
# freckles + grin
|
||||
for fx, fy in [(-10, -62), (10, -62), (-14, -58), (14, -58)]:
|
||||
d.point((cx + fx, cy + fy), fill=(180, 120, 80))
|
||||
# chest emblem
|
||||
d.polygon([(cx, cy - 20), (cx - 14, cy + 5), (cx + 14, cy + 5)], fill=(250, 210, 60))
|
||||
return img
|
||||
|
||||
|
||||
def mecha_dragon():
|
||||
w, h = 1344, 768
|
||||
img = city_backdrop(w, h, (8, 8, 20), (40, 20, 50), seed=23)
|
||||
d = ImageDraw.Draw(img)
|
||||
cx, cy = int(w * 0.6), int(h * 0.45)
|
||||
# body silhouette
|
||||
d.polygon(
|
||||
[(cx - 220, h), (cx - 120, cy + 60), (cx - 40, cy - 60), (cx + 60, cy - 140),
|
||||
(cx + 180, cy - 100), (cx + 240, cy + 40), (cx + 320, h)],
|
||||
fill=(10, 10, 14),
|
||||
)
|
||||
# head / jaws
|
||||
d.polygon(
|
||||
[(cx + 60, cy - 140), (cx + 10, cy - 220), (cx + 120, cy - 190), (cx + 180, cy - 100)],
|
||||
fill=(14, 14, 18),
|
||||
)
|
||||
d.polygon([(cx + 20, cy - 210), (cx - 40, cy - 250), (cx + 40, cy - 225)], fill=(16, 16, 20))
|
||||
# glowing eyes and chest core
|
||||
d.ellipse([cx + 60, cy - 195, cx + 80, cy - 180], fill=(255, 40, 40))
|
||||
d.ellipse([cx + 95, cy - 190, cx + 112, cy - 176], fill=(255, 40, 40))
|
||||
core = Image.new("RGB", (w, h))
|
||||
cd = ImageDraw.Draw(core)
|
||||
cd.ellipse([cx - 20, cy - 40, cx + 60, cy + 40], fill=(255, 60, 60))
|
||||
core = core.filter(ImageFilter.GaussianBlur(18))
|
||||
img.paste(Image.blend(img.crop((0, 0, w, h)), core, 0.5), (0, 0), core.convert("L"))
|
||||
# lightning bolts
|
||||
rng = random.Random(5)
|
||||
for _ in range(4):
|
||||
x0 = rng.randint(cx - 100, cx + 200)
|
||||
y0 = 0
|
||||
pts = [(x0, y0)]
|
||||
while pts[-1][1] < cy - 160:
|
||||
px, py = pts[-1]
|
||||
pts.append((px + rng.randint(-40, 40), py + rng.randint(30, 70)))
|
||||
d.line(pts, fill=(120, 180, 255), width=4)
|
||||
return img
|
||||
|
||||
|
||||
def bench_ref():
|
||||
img = city_backdrop(1344, 768, (200, 120, 60), (240, 200, 140), seed=42)
|
||||
return img
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
superboy().save(f"{OUT}/red_superboy_on_city_roof.png")
|
||||
mecha_dragon().save(f"{OUT}/mecha_dragon_lightning.png")
|
||||
bench_ref().save(f"{OUT}/bench_ref_1344x768.png")
|
||||
print("wrote 3 images to", OUT)
|
||||
Executable
+16
@@ -0,0 +1,16 @@
|
||||
#!/bin/bash
|
||||
# Start ComfyUI headless on the pod inside tmux, listening on 127.0.0.1:8188.
|
||||
# Access from your laptop via scripts/local/port-forward.sh
|
||||
set -euo pipefail
|
||||
|
||||
SESSION=comfyui
|
||||
LOG=/workspace/minimax-h3/comfyui.log
|
||||
|
||||
if tmux has-session -t "$SESSION" 2>/dev/null; then
|
||||
echo "ComfyUI tmux session already running"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
tmux new -s "$SESSION" -d \
|
||||
"cd /ComfyUI && python main.py --listen 127.0.0.1 --port 8188 2>&1 | tee -a $LOG"
|
||||
echo "started; tail $LOG"
|
||||
Reference in New Issue
Block a user