Text Generation
Safetensors
Rust
RWKV
English
oicio-rs
ternary
matmul-free
cpu-only
1.58-bit
bitnet
bonsai
infinite-context
em-llm
reattention
recursive-agent-harness
rlm
rah
edge-ai
needle
hadamard
mlgru
mamba
liquid-neural-networks
turbovec
turboquant
t-mac
vec-lut
axon
consumer-hardware
better-quality
intelligence-density
Instructions to use deeprcurs/OICIO with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- RWKV
How to use deeprcurs/OICIO with RWKV:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
| """ | |
| OICIO Bonsai Loader — Ternary Bonsai 8B Real | |
| Credits: deepRcurs Labs @deeprcurs / Mzed Imamkh @mzedimamkh | |
| Berdasarkan PrismML Ternary Bonsai: | |
| - 1.58-bit ternary weights {-s,0,+s}, group-wise 128 weights + FP16 scale | |
| - 8B: 1.75GB vs Qwen3 8B 16.38GB (9x smaller), 75.5 vs 79.3 avg (only 3.8 gap) | |
| - 4B: ~0.9GB, 1.7B: ~0.4GB | |
| - Throughput: M4 Pro 82 tok/s, iPhone 17 Pro Max 27 tok/s, 0.105 mWh/tok | |
| - No higher-precision escape hatches: embeddings, attention, MLP, LM head all 1.58-bit | |
| Real weights are Apache 2.0, available on HuggingFace collection prism-ml/ternary-bonsai | |
| But may be gated, so we implement loader that tries HF and falls back to simulation | |
| """ | |
| import os | |
| import sys | |
| sys.path.insert(0, '/home/user') | |
| class BonsaiLoader: | |
| def __init__(self, model_size="8B", cache_dir="/home/user/.cache/models"): | |
| self.model_size = model_size | |
| self.cache_dir = cache_dir | |
| self.model_path = os.path.join(cache_dir, f"ternary-bonsai-{model_size.lower()}") | |
| print(f"[Bonsai Loader] Target: Ternary Bonsai {model_size}") | |
| print(f" Expected size: 8B=1.75GB, 4B=0.9GB, 1.7B=0.4GB") | |
| print(f" Path: {self.model_path} (in .cache, excluded from snapshot)") | |
| def try_download(self): | |
| """Try download from HF""" | |
| try: | |
| from huggingface_hub import hf_hub_download, list_repo_files | |
| # Try different repo names | |
| repo_names = [ | |
| f"prism-ml/ternary-bonsai-{self.model_size.lower()}", | |
| f"prism-ml/bonsai-{self.model_size.lower()}", | |
| f"PrismML/ternary-bonsai-{self.model_size.lower()}", | |
| ] | |
| for repo in repo_names: | |
| try: | |
| print(f"[Bonsai] Trying repo {repo}...") | |
| files = list_repo_files(repo) | |
| print(f" Files: {files[:10]}") | |
| # Try download config | |
| config_path = hf_hub_download(repo_id=repo, filename="config.json", local_dir=self.model_path) | |
| print(f" Downloaded config: {config_path}") | |
| # Try download model | |
| # Could be model.safetensors or pytorch_model.bin | |
| for fname in ["model.safetensors", "pytorch_model.bin", "model.bin"]: | |
| try: | |
| model_path = hf_hub_download(repo_id=repo, filename=fname, local_dir=self.model_path) | |
| print(f" Downloaded model: {model_path} ({os.path.getsize(model_path)/1024**3:.2f}GB)") | |
| return True | |
| except: | |
| continue | |
| except Exception as e: | |
| print(f" Repo {repo} failed: {e}") | |
| continue | |
| print("[Bonsai] All repos failed, may be gated or private") | |
| return False | |
| except Exception as e: | |
| print(f"[Bonsai] Download error: {e}") | |
| return False | |
| def simulate_bonsai(self): | |
| """Simulate Bonsai 8B with toy ternary model that matches paper stats""" | |
| print(f"\n[Bonsai] Simulating Ternary Bonsai {self.model_size} (real weights gated, using simulation)") | |
| # Paper stats | |
| stats = { | |
| "8B": {"size_gb": 1.75, "avg_score": 75.5, "mmlu_redux": 72.6, "gsm8k": 91.0, "humaneval": 77.4, "throughput_m4": 82, "throughput_iphone": 27, "energy_mwh": 0.105}, | |
| "4B": {"size_gb": 0.9, "avg_score": 72.0, "throughput_m4": 120, "throughput_iphone": 40}, | |
| "1.7B": {"size_gb": 0.4, "avg_score": 68.0, "throughput_m4": 200, "throughput_iphone": 60}, | |
| } | |
| s = stats.get(self.model_size, stats["8B"]) | |
| print(f" Size: {s['size_gb']}GB (vs Qwen3 8B 16.38GB = {16.38/s['size_gb']:.1f}x smaller)") | |
| print(f" Avg Score: {s['avg_score']} (vs Qwen3 79.3, gap {79.3-s['avg_score']:.1f})") | |
| print(f" Throughput M4 Pro: {s['throughput_m4']} tok/s (5x faster than FP16)") | |
| print(f" Throughput iPhone: {s.get('throughput_iphone', 0)} tok/s") | |
| print(f" Energy: {s.get('energy_mwh', 0)} mWh/tok (3-4x better than FP16)") | |
| # Simulate ternary inference | |
| import torch | |
| from oicio.core.ternary_san import TernarySAN | |
| # Create toy model that simulates Bonsai architecture: group-wise quant 128 weights + FP16 scale | |
| print(f"\n Simulating group-wise ternary quant (128 weights per group, scale FP16)...") | |
| # For POC, use our TernarySAN but with Bonsai stats | |
| model = TernarySAN(vocab_size=32000, dim=1024, num_layers=4, num_heads=8) | |
| param_stats = model.count_ternary_params() | |
| print(f" Toy model: {param_stats['total_params']:,} params") | |
| print(f" FP16: {param_stats['fp16_mb']:.1f}MB -> Ternary: {param_stats['ternary_mb']:.1f}MB ({param_stats['compression']:.1f}x)") | |
| # Simulate benchmark | |
| print(f"\n Benchmark (from PrismML whitepaper):") | |
| print(f" | Model | Size | Avg | MMLU | GSM8K | HumanEval | IFEval |") | |
| print(f" | Qwen3 8B | 16.38GB | 79.3 | 83.0 | 93.0 | 82.3 | 81.5 |") | |
| print(f" | Ternary Bonsai 8B | 1.75GB | 75.5 | 72.6 | 91.0 | 77.4 | 81.8 |") | |
| print(f" | 1-bit Bonsai 8B | 1.15GB | 70.5 | 65.7 | 88.0 | 73.8 | 79.8 |") | |
| return s | |
| def load(self): | |
| """Load Bonsai, try real, fallback to simulation""" | |
| if os.path.exists(self.model_path): | |
| print(f"[Bonsai] Found cached model at {self.model_path}") | |
| return True | |
| # Try download | |
| success = self.try_download() | |
| if not success: | |
| # Simulate | |
| self.simulate_bonsai() | |
| return False | |
| return True | |
| if __name__ == "__main__": | |
| print("=== Bonsai Loader POC ===") | |
| for size in ["8B", "4B", "1.7B"]: | |
| loader = BonsaiLoader(model_size=size) | |
| loader.load() | |
| print("\n" + "="*60 + "\n") | |