Instructions to use yitongl/minimax-h3-nvfp4-lambda-modality with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MiniMax H3
How to use yitongl/minimax-h3-nvfp4-lambda-modality with MiniMax H3:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
| #!/usr/bin/env python | |
| """One grid video over the lambda experiments, each panel captioned with what it is. | |
| Composed here rather than with an ffmpeg filtergraph because the caption is the whole point -- six | |
| unlabelled panels are unreadable, and `drawtext` needs a freetype-enabled build that the bundled | |
| imageio-ffmpeg binary does not reliably have. PIL draws the label, imageio writes the file. | |
| Panels are scaled to a common height and letterboxed to a common width, so a canvas difference | |
| between runs shows as padding instead of silently stretching one panel against the others. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| from pathlib import Path | |
| import av | |
| import numpy as np | |
| from PIL import Image, ImageDraw | |
| PANELS = [ | |
| ("0_bf16", "BF16 reference"), | |
| ("1_plain_w4a4", "1. plain W4A4 (lambda=1, no LoRA)"), | |
| ("2_lambda_none", "2. lambda=1 + rank-32 LoRA"), | |
| ("3_lambda_all", "3. lambda from all tokens"), | |
| ("4_lambda_video", "4. lambda from video tokens"), | |
| ("5_lambda_text", "5. lambda from text tokens"), | |
| ] | |
| def read(path: Path, height: int) -> list[np.ndarray]: | |
| c = av.open(str(path)) | |
| v = c.streams.video[0] | |
| v.thread_type, v.thread_count = "NONE", 1 | |
| out = [] | |
| for f in c.decode(video=0): | |
| a = f.to_ndarray(format="rgb24") | |
| w = int(round(a.shape[1] * height / a.shape[0] / 2)) * 2 | |
| out.append(np.asarray(Image.fromarray(a).resize((w, height), Image.BICUBIC))) | |
| c.close() | |
| return out | |
| def label(frame: np.ndarray, text: str) -> np.ndarray: | |
| im = Image.fromarray(frame) | |
| d = ImageDraw.Draw(im, "RGBA") | |
| d.rectangle([0, 0, im.width, 20], fill=(0, 0, 0, 170)) | |
| d.text((6, 5), text, fill=(255, 255, 255)) | |
| return np.asarray(im) | |
| def main() -> int: | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--dir", default="out/lambda_exps") | |
| ap.add_argument("--out", default="out/lambda_exps/grid.mp4") | |
| ap.add_argument("--height", type=int, default=300) | |
| ap.add_argument("--fps", type=int, default=24) | |
| args = ap.parse_args() | |
| d = Path(args.dir) | |
| have = [(s, t) for s, t in PANELS if (d / f"{s}.mp4").exists()] | |
| if len(have) < 2: | |
| print(f"only {len(have)} panels present, nothing to stack") | |
| return 0 | |
| clips = [read(d / f"{s}.mp4", args.height) for s, _ in have] | |
| n = min(len(c) for c in clips) | |
| w = max(c[0].shape[1] for c in clips) | |
| cols = (len(have) + 1) // 2 | |
| import imageio.v2 as imageio | |
| wr = imageio.get_writer(args.out, fps=args.fps, codec="libx264", quality=8, | |
| macro_block_size=2) | |
| for i in range(n): | |
| tiles = [] | |
| for (s, t), c in zip(have, clips): | |
| f = label(c[i], t) | |
| if f.shape[1] < w: # letterbox, never stretch | |
| pad = np.zeros((f.shape[0], w, 3), np.uint8) | |
| o = (w - f.shape[1]) // 2 | |
| pad[:, o:o + f.shape[1]] = f | |
| f = pad | |
| tiles.append(f) | |
| while len(tiles) % cols: | |
| tiles.append(np.zeros_like(tiles[0])) | |
| rows = [np.concatenate(tiles[k:k + cols], axis=1) for k in range(0, len(tiles), cols)] | |
| wr.append_data(np.concatenate(rows, axis=0)) | |
| wr.close() | |
| print(f"wrote {args.out}: {len(have)} panels, {n} frames, {cols} cols") | |
| for s, t in have: | |
| print(f" {s:<16} {t}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |