Invalid JSON:Unexpected non-whitespace character after JSONat line 78, column 2
| { | |
| "model": "North-ML1/Aurora-Proelia-ChatML", | |
| "before_checkpoint": "North-ML1/Aurora-Proelia", | |
| "after_checkpoint": "/home/arthur/ember-proelia/exports/aurora-proelia-chatml-20260815", | |
| "matched_public_slice": { | |
| "mmlu": { | |
| "released": { | |
| "dataset": "cais/mmlu", | |
| "config": "all", | |
| "split": "test", | |
| "n": 57, | |
| "correct": 14, | |
| "accuracy": 0.24561403508771928 | |
| }, | |
| "chatml": { | |
| "dataset": "cais/mmlu", | |
| "config": "all", | |
| "split": "test", | |
| "n": 57, | |
| "correct": 16, | |
| "accuracy": 0.2807017543859649 | |
| } | |
| }, | |
| "arc_challenge": { | |
| "released": { | |
| "dataset": "allenai/ai2_arc", | |
| "config": "ARC-Challenge", | |
| "split": "test", | |
| "n": 50, | |
| "correct": 13, | |
| "accuracy": 0.26 | |
| }, | |
| "chatml": { | |
| "dataset": "allenai/ai2_arc", | |
| "config": "ARC-Challenge", | |
| "split": "test", | |
| "n": 50, | |
| "correct": 15, | |
| "accuracy": 0.3 | |
| } | |
| }, | |
| "hellaswag": { | |
| "released": { | |
| "dataset": "Rowan/hellaswag", | |
| "split": "validation", | |
| "n": 50, | |
| "correct": 19, | |
| "accuracy": 0.38 | |
| }, | |
| "chatml": { | |
| "dataset": "Rowan/hellaswag", | |
| "split": "validation", | |
| "n": 50, | |
| "correct": 19, | |
| "accuracy": 0.38 | |
| } | |
| }, | |
| "gsm8k": { | |
| "released": { | |
| "dataset": "openai/gsm8k", | |
| "config": "main", | |
| "split": "test", | |
| "n": 50, | |
| "correct": 1, | |
| "accuracy": 0.02 | |
| }, | |
| "chatml": { | |
| "dataset": "openai/gsm8k", | |
| "config": "main", | |
| "split": "test", | |
| "n": 50, | |
| "correct": 0, | |
| "accuracy": 0.0 | |
| } | |
| } | |
| }, | |
| "note": "Same public dataset slices and native Aurora scoring code; directional regression check, not an official leaderboard evaluation." | |
| }\n |