File size: 1,130 Bytes
c34f594
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
model: Macaron-V1-Coding-Venti
source:
  name: Model Card
  url: https://huggingface.co/mindlab-research/Macaron-V1-Coding-Venti
notes:
  higher_is_better: true
  missing_values_are_null: false
  checkpoint_type: merged GLM-5.2 BF16 checkpoint with Macaron-V1 L2 Coding LoRA update merged into the base weights
baselines:
  - Macaron V1
  - GLM 5.2
  - GPT 5.5
  - Claude Opus 4.8
  - Gemini 3.1 Pro
  - Qwen 3.7 Max
  - Minimax M3
results:
  - benchmark: SWE Verified
    scores: {Macaron V1: 85.6, GLM 5.2: 80.4, GPT 5.5: 82.9, Claude Opus 4.8: 88.6, Gemini 3.1 Pro: 80.6, Qwen 3.7 Max: 80.4, Minimax M3: 80.5}
  - benchmark: TerminalBench 2.1
    scores: {Macaron V1: 87.6, GLM 5.2: 82.7, GPT 5.5: 83.4, Claude Opus 4.8: 78.9, Gemini 3.1 Pro: 70.7, Qwen 3.7 Max: 73.5, Minimax M3: 66.0}
  - benchmark: DeepSWE
    scores: {Macaron V1: 58.4, GLM 5.2: 54.9, GPT 5.5: 70.0, Claude Opus 4.8: 58.0, Gemini 3.1 Pro: 10.0, Qwen 3.7 Max: 18.0, Minimax M3: 20.0}
  - benchmark: SWE Atlas QnA
    scores: {Macaron V1: 49.5, GLM 5.2: 48.9, GPT 5.5: 45.4, Claude Opus 4.8: 57.3, Gemini 3.1 Pro: 13.5, Qwen 3.7 Max: 22.6, Minimax M3: 37.9}