MAMBA_7B / src /build_full_report.py
namanadep's picture
Upload src/build_full_report.py with huggingface_hub
8f7ce71 verified
Raw
History Blame Contribute Delete
5.3 kB
import json
json_file = "/home/adminuser/mamba_vs_qwen_results.json"
out_md = "/home/adminuser/CODESTRAL_MAMBA_7B_VS_QWEN_7B_COMPARISON.md"
with open(json_file, "r", encoding="utf-8") as f:
data = json.load(f)
lines = []
lines.append("# Exhaustive Technical Benchmark & Architectural Comparison: Codestral Mamba 7B vs. Qwen 2.5 7B")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## 1. Executive Summary")
lines.append("")
lines.append("This document provides an exhaustive empirical and theoretical evaluation comparing **Codestral Mamba 7B** (`mamba-codestral:7b`), a specialized **Selective State Space Model (SSM)** by Mistral AI, against **Qwen 2.5 7B** (`qwen2.5:7b-instruct`), a state-of-the-art **Transformer-based** foundation model by Alibaba Cloud.")
lines.append("")
lines.append("Both models were deployed locally via **Ollama** on **NVIDIA H200 NVL GPUs** and evaluated across 10 rigorous technical benchmarks covering lock-free data structures, C++ memory safety & race conditions, Linux eBPF kernel tracing, async WebSockets, Rust zero-copy parsing, and State Space Model mechanics.")
lines.append("")
lines.append("### Key Empirical Findings")
lines.append("- **Inference Latency**: Codestral Mamba 7B achieves an average latency of **4.28 seconds** per query vs. Qwen 2.5 7B's **7.40 seconds** (**42.2% faster completion time**).")
lines.append("- **Hardware Generation Speed**: Both models output tokens at virtually identical hardware limits (**~194.8 tokens/sec** for Mamba vs. **~193.3 tokens/sec** for Qwen 2.5 on H200 GPUs).")
lines.append("- **Content & Explanation Density**: Qwen 2.5 7B generates **2.25x more content** on average (4,347 characters vs. 1,930 characters), delivering comprehensive docstrings, edge-case coverage, and structural explanations.")
lines.append("- **Memory Footprint**: Mamba operates with a constant $O(1)$ recurrent memory state, avoiding the exponential $O(N)$ KV-cache VRAM expansion of Transformers during extreme long-context processing (up to 256k tokens).")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## 2. Quantitative Benchmark Performance Table")
lines.append("")
lines.append("| Test ID | Category | Technical Benchmark Title | Mamba Latency (s) | Qwen Latency (s) | Mamba Speed (t/s) | Qwen Speed (t/s) | Mamba Output Chars | Qwen Output Chars |")
lines.append("| :---: | :--- | :--- | :---: | :---: | :---: | :---: | :---: | :---: |")
tot_m_t, tot_q_t, tot_m_c, tot_q_c = 0, 0, 0, 0
for item in data:
tid = item["id"]
cat = item["category"]
title = item["title"]
m = item["mamba"]
q = item["qwen"]
tot_m_t += m["time_sec"]
tot_q_t += q["time_sec"]
tot_m_c += m["char_len"]
tot_q_c += q["char_len"]
lines.append(f"| {tid:02d} | {cat} | {title} | {m['time_sec']:.2f}s | {q['time_sec']:.2f}s | {m['tps']} | {q['tps']} | {m['char_len']} | {q['char_len']} |")
lines.append("")
lines.append("**Summary Averages**:")
lines.append(f"- **Average Response Time**: Codestral Mamba: **{tot_m_t/len(data):.2f}s** | Qwen 2.5: **{tot_q_t/len(data):.2f}s**")
lines.append(f"- **Average Response Length**: Codestral Mamba: **{tot_m_c/len(data):.0f} characters** | Qwen 2.5: **{tot_q_c/len(data):.0f} characters**")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## 3. In-Depth Side-by-Side Prompt Evaluations")
lines.append("")
for item in data:
tid = item["id"]
title = item["title"]
cat = item["category"]
prompt = item["prompt"]
m = item["mamba"]
q = item["qwen"]
lines.append(f"### Benchmark Test {tid:02d}: {title}")
lines.append(f"**Category**: `{cat}`")
lines.append("")
lines.append("**Prompt**:")
lines.append("```")
lines.append(prompt)
lines.append("```")
lines.append("")
lines.append(f"#### 🟢 Codestral Mamba 7B Response (`{m['time_sec']}s` | `{m['char_len']} chars` | `{m['tps']} t/s`)")
lines.append("```python")
lines.append(m['text'])
lines.append("```")
lines.append("")
lines.append(f"#### 🔵 Qwen 2.5 7B Instruct Response (`{q['time_sec']}s` | `{q['char_len']} chars` | `{q['tps']} t/s`)")
lines.append("```python")
lines.append(q['text'])
lines.append("```")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## 4. Architectural Analysis & Deployment Recommendations")
lines.append("")
lines.append("### Codestral Mamba 7B")
lines.append("- **Best For**: Real-time IDE inline autocomplete, long-context repository processing (256k tokens), streaming code microservices.")
lines.append("- **Pros**: Low latency, constant memory state $O(1)$, zero KV-cache overhead.")
lines.append("- **Cons**: Minimal inline documentation and concise explanations.")
lines.append("")
lines.append("### Qwen 2.5 7B Instruct")
lines.append("- **Best For**: Full-stack application design, code refactoring, complex bug fixing, unit test suite generation.")
lines.append("- **Pros**: Highly detailed, exhaustive docstrings, complete edge-case handling.")
lines.append("- **Cons**: Transformer KV-cache memory expansion at massive context windows.")
with open(out_md, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
print("Successfully wrote full exhaustive Markdown report to:", out_md)