File size: 4,385 Bytes
4f785fa
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
#!/usr/bin/env python3
"""
Sixpert K2 - Quick Benchmark Script
====================================
Runs basic performance benchmarks for Sixpert K2 (MoE).

Key advantage: Despite 8.9B total parameters, only ~1.2B are active
per token, making inference faster than dense models of similar size.

Usage:
    python benchmark.py --model SixpertK2.gguf
"""

import argparse
import time
import sys

try:
    from llama_cpp import Llama
except ImportError:
    print("Installing llama-cpp-python...")
    import subprocess
    subprocess.check_call([sys.executable, "-m", "pip", "install", "llama-cpp-python"])
    from llama_cpp import Llama


def benchmark_generation(model_path: str, tokens: int = 512):
    """Benchmark token generation speed."""
    print("\n=== Generation Benchmark ===")
    print(f"Generating {tokens} tokens...\n")

    llm = Llama(
        model_path=model_path,
        n_ctx=4096,
        n_gpu_layers=-1,
        verbose=False,
    )

    start = time.time()
    output = llm(
        "<|im_start|>user\nWrite a detailed essay about artificial intelligence and its impact on society.<|im_end|>\n<|im_start|>assistant\n",
        max_tokens=tokens,
        temperature=0.6,
        stream=False,
    )
    elapsed = time.time() - start

    tokens_per_sec = tokens / elapsed

    print(f"Generated: {tokens} tokens")
    print(f"Time: {elapsed:.2f}s")
    print(f"Speed: {tokens_per_sec:.1f} tokens/sec")
    print(f"Note: Only ~1.2B active params per token (MoE advantage)")


def benchmark_context(model_path: str, context_length: int = 16384):
    """Benchmark long-context processing speed."""
    print(f"\n=== Long-Context Benchmark ===")
    print(f"Processing {context_length} token context...\n")

    llm = Llama(
        model_path=model_path,
        n_ctx=context_length + 512,
        n_gpu_layers=-1,
        verbose=False,
    )

    # Create a long context prompt
    filler = "The evolution of artificial intelligence has been marked by several key milestones. " * (context_length // 15)
    prompt = f"<|im_start|>user\n{filler}\nBased on the above text, what are the main themes discussed?<|im_end|>\n<|im_start|>assistant\n"

    start = time.time()
    output = llm(prompt, max_tokens=200, stream=False)
    elapsed = time.time() - start

    prompt_tokens = output["usage"]["prompt_eval_count"]
    eval_time = output["usage"].get("prompt_eval_time", 1000) / 1000

    print(f"Context tokens: {prompt_tokens}")
    print(f"Processing time: {eval_time:.2f}s")
    print(f"Speed: {prompt_tokens / eval_time:.1f} tokens/sec")


def benchmark_reasoning(model_path: str):
    """Benchmark deep reasoning capability."""
    print(f"\n=== Deep Reasoning Benchmark ===")
    print(f"Testing multi-step reasoning...\n")

    llm = Llama(
        model_path=model_path,
        n_ctx=8192,
        n_gpu_layers=-1,
        verbose=False,
    )

    start = time.time()
    output = llm(
        "<|im_start|>user\nProve that there are infinitely many prime numbers. Provide a complete, rigorous mathematical proof.<|im_end|>\n<|im_start|>assistant\n",
        max_tokens=2048,
        temperature=0.3,
        stream=False,
    )
    elapsed = time.time() - start

    response_text = output["choices"][0]["text"]
    print(f"Response length: {len(response_text)} chars")
    print(f"Time: {elapsed:.2f}s")
    print(f"Tokens/sec: {2048 / elapsed:.1f}")
    print(f"\nFirst 200 chars: {response_text[:200]}...")


def main():
    parser = argparse.ArgumentParser(description="Sixpert K2 Benchmark")
    parser.add_argument("--model", type=str, default="SixpertK2.gguf", help="Path to GGUF model")
    parser.add_argument("--gen-tokens", type=int, default=512, help="Generation benchmark tokens")
    parser.add_argument("--ctx-length", type=int, default=16384, help="Context benchmark length")
    parser.add_argument("--all", action="store_true", help="Run all benchmarks")

    args = parser.parse_args()

    print("=" * 60)
    print("  Sixpert K2 Benchmark Suite")
    print("  Deep Reasoning Engine (MoE)")
    print("  Total: ~8.9B | Active: ~1.2B/token")
    print("=" * 60)

    benchmark_generation(args.model, args.gen_tokens)
    benchmark_context(args.model, args.ctx_length)
    benchmark_reasoning(args.model)

    print("\n" + "=" * 60)
    print("  Benchmark complete!")
    print("=" * 60)


if __name__ == "__main__":
    main()