Instructions to use MrMofer/MiniMax-H3-MLX-8bit with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use MrMofer/MiniMax-H3-MLX-8bit with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] huggingface-cli download --local-dir MiniMax-H3-MLX-8bit MrMofer/MiniMax-H3-MLX-8bit
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
| import json, struct | |
| import numpy as np | |
| import mlx.core as mx | |
| def _load(path, key): | |
| with open(path, 'rb') as f: | |
| n = struct.unpack('<Q', f.read(8))[0] | |
| hdr = json.loads(f.read(n)) | |
| e = hdr[key] | |
| start, end = e['data_offsets'] | |
| f.seek(8 + n + start) | |
| raw = f.read(end - start) | |
| return e, raw | |
| def get_tensor(path, key, dtype=None): | |
| """Bit-accurate tensor reader. BF16 is bit-reinterpreted (uint16 bits -> bf16).""" | |
| e, raw = _load(path, key) | |
| t = e['dtype'] | |
| if t == 'BF16': | |
| u16 = np.frombuffer(raw, dtype=np.uint16) | |
| f32 = np.left_shift(u16.astype(np.uint32), 16).view(np.float32).copy() | |
| a = mx.array(f32) | |
| if dtype in (None, 'bf16'): | |
| # exact bf16 round-trip: bf16 -> f32 is lossless | |
| return a.reshape(e['shape']) if dtype is None else a.astype(mx.bfloat16).reshape(e['shape']) | |
| return a.reshape(e['shape']) | |
| if t == 'F16': | |
| return mx.array(np.frombuffer(raw, dtype=np.float16).copy()).reshape(e['shape']) | |
| if t == 'F32': | |
| return mx.array(np.frombuffer(raw, dtype=np.float32).copy()).reshape(e['shape']) | |
| if t in ('U32', 'I32'): | |
| return mx.array(np.frombuffer(raw, dtype=np.uint32).copy()).reshape(e['shape']) | |
| raise ValueError(t) | |
| def header(path): | |
| with open(path, 'rb') as f: | |
| n = struct.unpack('<Q', f.read(8))[0] | |
| return json.loads(f.read(n)) | |