Submitting LFM2.5 models results.

#7
by FlameF0X - opened
Files changed (1) hide show
  1. models.json +1060 -0
models.json CHANGED
@@ -61,6 +61,1066 @@
61
  ],
62
  "models": [
63
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64
  "id": "liquidai-lfm2.5-230m",
65
  "name": "LiquidAI/LFM2.5-230M",
66
  "org": "LiquidAI",
 
61
  ],
62
  "models": [
63
  {
64
+ "id": "liquidai-lfm2.5-1.2b-base",
65
+ "name": "LiquidAI/LFM2.5-1.2B-Base",
66
+ "org": "LiquidAI",
67
+ "params_b": null,
68
+ "license": null,
69
+ "architecture": null,
70
+ "url": "https://huggingface.co/LiquidAI/LFM2.5-1.2B-Base",
71
+ "model_revision": "f6a5d174bc3e52bd0df245d69133f9930b4828d8",
72
+ "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
73
+ "runs": {
74
+ "bench-effortless-6-2026": {
75
+ "score": 0.0,
76
+ "n": 240,
77
+ "notes": "Exact-match, normalized.",
78
+ "categories": {
79
+ "Commonsense-reasoning": {
80
+ "n": 37,
81
+ "exact_match": 0.0
82
+ },
83
+ "Knowledge-basic": {
84
+ "n": 42,
85
+ "exact_match": 0.0
86
+ },
87
+ "Language-comprehension": {
88
+ "n": 42,
89
+ "exact_match": 0.0
90
+ },
91
+ "Logic-deduction": {
92
+ "n": 42,
93
+ "exact_match": 0.0
94
+ },
95
+ "Math-arithmetic": {
96
+ "n": 40,
97
+ "exact_match": 0.0
98
+ },
99
+ "Pattern-recognition": {
100
+ "n": 37,
101
+ "exact_match": 0.0
102
+ }
103
+ }
104
+ },
105
+ "bench-easy-6-2026": {
106
+ "score": 0.3087,
107
+ "n": 238,
108
+ "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
109
+ "categories": {
110
+ "Commonsense-causality": {
111
+ "n": 10,
112
+ "hybrid_score": 0.7338
113
+ },
114
+ "Commonsense-reasoning": {
115
+ "n": 10,
116
+ "hybrid_score": 0.7347
117
+ },
118
+ "Commonsense-simulation": {
119
+ "n": 10,
120
+ "hybrid_score": 0.742
121
+ },
122
+ "Knowledge-basic": {
123
+ "n": 33,
124
+ "hybrid_score": 0.0303
125
+ },
126
+ "Knowledge-definitions": {
127
+ "n": 38,
128
+ "hybrid_score": 0.8057
129
+ },
130
+ "Language-comprehension": {
131
+ "n": 10,
132
+ "hybrid_score": 0.7448
133
+ },
134
+ "Language-structure": {
135
+ "n": 10,
136
+ "hybrid_score": 0.2311
137
+ },
138
+ "Language-transformation": {
139
+ "n": 10,
140
+ "hybrid_score": 0.7685
141
+ },
142
+ "Logic-consistency": {
143
+ "n": 10,
144
+ "hybrid_score": 0.0
145
+ },
146
+ "Logic-deduction": {
147
+ "n": 15,
148
+ "hybrid_score": 0.0667
149
+ },
150
+ "Logic-pattern": {
151
+ "n": 10,
152
+ "hybrid_score": 0.0
153
+ },
154
+ "Math-arithmetic": {
155
+ "n": 33,
156
+ "hybrid_score": 0.0
157
+ },
158
+ "Math-pattern": {
159
+ "n": 14,
160
+ "hybrid_score": 0.0
161
+ },
162
+ "Math-reasoning": {
163
+ "n": 15,
164
+ "hybrid_score": 0.0868
165
+ },
166
+ "Pattern-matching": {
167
+ "n": 10,
168
+ "hybrid_score": 0.0
169
+ }
170
+ }
171
+ },
172
+ "bench-mid-6-2026": {
173
+ "score": 0.621,
174
+ "n": 143,
175
+ "acc": 0.5385,
176
+ "acc_norm": 0.6154,
177
+ "soft_score": 0.5455,
178
+ "soft_score_norm": 0.621,
179
+ "stderr": 0.0404,
180
+ "categories": {
181
+ "Commonsense-causality": {
182
+ "n": 5,
183
+ "acc": 0.6,
184
+ "acc_norm": 0.6
185
+ },
186
+ "Commonsense-reasoning": {
187
+ "n": 10,
188
+ "acc": 0.6,
189
+ "acc_norm": 0.5
190
+ },
191
+ "Commonsense-simulation": {
192
+ "n": 10,
193
+ "acc": 0.4,
194
+ "acc_norm": 0.7
195
+ },
196
+ "Knowledge-basic": {
197
+ "n": 7,
198
+ "acc": 0.5714,
199
+ "acc_norm": 0.5714
200
+ },
201
+ "Knowledge-definitions": {
202
+ "n": 10,
203
+ "acc": 0.3,
204
+ "acc_norm": 0.9
205
+ },
206
+ "Language-comprehension": {
207
+ "n": 10,
208
+ "acc": 0.6,
209
+ "acc_norm": 0.7
210
+ },
211
+ "Language-structure": {
212
+ "n": 10,
213
+ "acc": 0.5,
214
+ "acc_norm": 0.4
215
+ },
216
+ "Language-transformation": {
217
+ "n": 10,
218
+ "acc": 0.5,
219
+ "acc_norm": 0.7
220
+ },
221
+ "Logic-consistency": {
222
+ "n": 5,
223
+ "acc": 0.0,
224
+ "acc_norm": 0.0
225
+ },
226
+ "Logic-deduction": {
227
+ "n": 10,
228
+ "acc": 0.3,
229
+ "acc_norm": 0.3
230
+ },
231
+ "Logic-pattern": {
232
+ "n": 10,
233
+ "acc": 0.5,
234
+ "acc_norm": 0.6
235
+ },
236
+ "Math-arithmetic": {
237
+ "n": 8,
238
+ "acc": 0.875,
239
+ "acc_norm": 0.875
240
+ },
241
+ "Math-pattern": {
242
+ "n": 7,
243
+ "acc": 0.8571,
244
+ "acc_norm": 0.8571
245
+ },
246
+ "Math-reasoning": {
247
+ "n": 10,
248
+ "acc": 0.4,
249
+ "acc_norm": 0.3
250
+ },
251
+ "Pattern-generation": {
252
+ "n": 4,
253
+ "acc": 1.0,
254
+ "acc_norm": 1.0
255
+ },
256
+ "Pattern-matching": {
257
+ "n": 8,
258
+ "acc": 1.0,
259
+ "acc_norm": 0.875
260
+ },
261
+ "Pattern-recognition": {
262
+ "n": 9,
263
+ "acc": 0.4444,
264
+ "acc_norm": 0.6667
265
+ }
266
+ }
267
+ },
268
+ "bench-AGI": {
269
+ "score": null,
270
+ "n": null,
271
+ "notes": "Not yet evaluated on this tier."
272
+ }
273
+ }
274
+ },
275
+ {
276
+ "id": "liquidai-lfm2.5-1.2b-instruct",
277
+ "name": "LiquidAI/LFM2.5-1.2B-Instruct",
278
+ "org": "LiquidAI",
279
+ "params_b": null,
280
+ "license": null,
281
+ "architecture": null,
282
+ "url": "https://huggingface.co/LiquidAI/LFM2.5-1.2B-Instruct",
283
+ "model_revision": "868df74dd56ff8a0c2ac5dbf281690c2dbebe4c9",
284
+ "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
285
+ "runs": {
286
+ "bench-effortless-6-2026": {
287
+ "score": 0.1208,
288
+ "n": 240,
289
+ "notes": "Exact-match, normalized.",
290
+ "categories": {
291
+ "Commonsense-reasoning": {
292
+ "n": 37,
293
+ "exact_match": 0.0
294
+ },
295
+ "Knowledge-basic": {
296
+ "n": 42,
297
+ "exact_match": 0.119
298
+ },
299
+ "Language-comprehension": {
300
+ "n": 42,
301
+ "exact_match": 0.0
302
+ },
303
+ "Logic-deduction": {
304
+ "n": 42,
305
+ "exact_match": 0.0952
306
+ },
307
+ "Math-arithmetic": {
308
+ "n": 40,
309
+ "exact_match": 0.475
310
+ },
311
+ "Pattern-recognition": {
312
+ "n": 37,
313
+ "exact_match": 0.027
314
+ }
315
+ }
316
+ },
317
+ "bench-easy-6-2026": {
318
+ "score": 0.441,
319
+ "n": 238,
320
+ "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
321
+ "categories": {
322
+ "Commonsense-causality": {
323
+ "n": 10,
324
+ "hybrid_score": 0.7508
325
+ },
326
+ "Commonsense-reasoning": {
327
+ "n": 10,
328
+ "hybrid_score": 0.7711
329
+ },
330
+ "Commonsense-simulation": {
331
+ "n": 10,
332
+ "hybrid_score": 0.7765
333
+ },
334
+ "Knowledge-basic": {
335
+ "n": 33,
336
+ "hybrid_score": 0.303
337
+ },
338
+ "Knowledge-definitions": {
339
+ "n": 38,
340
+ "hybrid_score": 0.8172
341
+ },
342
+ "Language-comprehension": {
343
+ "n": 10,
344
+ "hybrid_score": 0.7459
345
+ },
346
+ "Language-structure": {
347
+ "n": 10,
348
+ "hybrid_score": 0.4632
349
+ },
350
+ "Language-transformation": {
351
+ "n": 10,
352
+ "hybrid_score": 0.7849
353
+ },
354
+ "Logic-consistency": {
355
+ "n": 10,
356
+ "hybrid_score": 0.0
357
+ },
358
+ "Logic-deduction": {
359
+ "n": 15,
360
+ "hybrid_score": 0.0667
361
+ },
362
+ "Logic-pattern": {
363
+ "n": 10,
364
+ "hybrid_score": 0.2
365
+ },
366
+ "Math-arithmetic": {
367
+ "n": 33,
368
+ "hybrid_score": 0.2727
369
+ },
370
+ "Math-pattern": {
371
+ "n": 14,
372
+ "hybrid_score": 0.0
373
+ },
374
+ "Math-reasoning": {
375
+ "n": 15,
376
+ "hybrid_score": 0.4654
377
+ },
378
+ "Pattern-matching": {
379
+ "n": 10,
380
+ "hybrid_score": 0.2
381
+ }
382
+ }
383
+ },
384
+ "bench-mid-6-2026": {
385
+ "score": 0.607,
386
+ "n": 143,
387
+ "acc": 0.5175,
388
+ "acc_norm": 0.6014,
389
+ "soft_score": 0.5231,
390
+ "soft_score_norm": 0.607,
391
+ "stderr": 0.0406,
392
+ "categories": {
393
+ "Commonsense-causality": {
394
+ "n": 5,
395
+ "acc": 0.8,
396
+ "acc_norm": 0.8
397
+ },
398
+ "Commonsense-reasoning": {
399
+ "n": 10,
400
+ "acc": 0.6,
401
+ "acc_norm": 0.5
402
+ },
403
+ "Commonsense-simulation": {
404
+ "n": 10,
405
+ "acc": 0.3,
406
+ "acc_norm": 0.6
407
+ },
408
+ "Knowledge-basic": {
409
+ "n": 7,
410
+ "acc": 1.0,
411
+ "acc_norm": 1.0
412
+ },
413
+ "Knowledge-definitions": {
414
+ "n": 10,
415
+ "acc": 0.4,
416
+ "acc_norm": 0.9
417
+ },
418
+ "Language-comprehension": {
419
+ "n": 10,
420
+ "acc": 0.6,
421
+ "acc_norm": 0.6
422
+ },
423
+ "Language-structure": {
424
+ "n": 10,
425
+ "acc": 0.2,
426
+ "acc_norm": 0.3
427
+ },
428
+ "Language-transformation": {
429
+ "n": 10,
430
+ "acc": 0.6,
431
+ "acc_norm": 0.6
432
+ },
433
+ "Logic-consistency": {
434
+ "n": 5,
435
+ "acc": 0.0,
436
+ "acc_norm": 0.2
437
+ },
438
+ "Logic-deduction": {
439
+ "n": 10,
440
+ "acc": 0.0,
441
+ "acc_norm": 0.3
442
+ },
443
+ "Logic-pattern": {
444
+ "n": 10,
445
+ "acc": 0.4,
446
+ "acc_norm": 0.3
447
+ },
448
+ "Math-arithmetic": {
449
+ "n": 8,
450
+ "acc": 1.0,
451
+ "acc_norm": 1.0
452
+ },
453
+ "Math-pattern": {
454
+ "n": 7,
455
+ "acc": 1.0,
456
+ "acc_norm": 1.0
457
+ },
458
+ "Math-reasoning": {
459
+ "n": 10,
460
+ "acc": 0.3,
461
+ "acc_norm": 0.3
462
+ },
463
+ "Pattern-generation": {
464
+ "n": 4,
465
+ "acc": 0.75,
466
+ "acc_norm": 0.75
467
+ },
468
+ "Pattern-matching": {
469
+ "n": 8,
470
+ "acc": 0.75,
471
+ "acc_norm": 0.75
472
+ },
473
+ "Pattern-recognition": {
474
+ "n": 9,
475
+ "acc": 0.5556,
476
+ "acc_norm": 0.6667
477
+ }
478
+ }
479
+ },
480
+ "bench-AGI": {
481
+ "score": null,
482
+ "n": null,
483
+ "notes": "Not yet evaluated on this tier."
484
+ }
485
+ }
486
+ },
487
+ {
488
+ "id": "liquidai-lfm2.5-230m-base",
489
+ "name": "LiquidAI/LFM2.5-230M-Base",
490
+ "org": "LiquidAI",
491
+ "params_b": null,
492
+ "license": null,
493
+ "architecture": null,
494
+ "url": "https://huggingface.co/LiquidAI/LFM2.5-230M-Base",
495
+ "model_revision": "9d2be5519834990d30996f878b6771cccbd24f2c",
496
+ "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
497
+ "runs": {
498
+ "bench-effortless-6-2026": {
499
+ "score": 0.0042,
500
+ "n": 240,
501
+ "notes": "Exact-match, normalized.",
502
+ "categories": {
503
+ "Commonsense-reasoning": {
504
+ "n": 37,
505
+ "exact_match": 0.0
506
+ },
507
+ "Knowledge-basic": {
508
+ "n": 42,
509
+ "exact_match": 0.0
510
+ },
511
+ "Language-comprehension": {
512
+ "n": 42,
513
+ "exact_match": 0.0
514
+ },
515
+ "Logic-deduction": {
516
+ "n": 42,
517
+ "exact_match": 0.0
518
+ },
519
+ "Math-arithmetic": {
520
+ "n": 40,
521
+ "exact_match": 0.0
522
+ },
523
+ "Pattern-recognition": {
524
+ "n": 37,
525
+ "exact_match": 0.027
526
+ }
527
+ }
528
+ },
529
+ "bench-easy-6-2026": {
530
+ "score": 0.2876,
531
+ "n": 238,
532
+ "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
533
+ "categories": {
534
+ "Commonsense-causality": {
535
+ "n": 10,
536
+ "hybrid_score": 0.7301
537
+ },
538
+ "Commonsense-reasoning": {
539
+ "n": 10,
540
+ "hybrid_score": 0.7137
541
+ },
542
+ "Commonsense-simulation": {
543
+ "n": 10,
544
+ "hybrid_score": 0.7434
545
+ },
546
+ "Knowledge-basic": {
547
+ "n": 33,
548
+ "hybrid_score": 0.0
549
+ },
550
+ "Knowledge-definitions": {
551
+ "n": 38,
552
+ "hybrid_score": 0.7888
553
+ },
554
+ "Language-comprehension": {
555
+ "n": 10,
556
+ "hybrid_score": 0.7182
557
+ },
558
+ "Language-structure": {
559
+ "n": 10,
560
+ "hybrid_score": 0.2974
561
+ },
562
+ "Language-transformation": {
563
+ "n": 10,
564
+ "hybrid_score": 0.6079
565
+ },
566
+ "Logic-consistency": {
567
+ "n": 10,
568
+ "hybrid_score": 0.0
569
+ },
570
+ "Logic-deduction": {
571
+ "n": 15,
572
+ "hybrid_score": 0.0
573
+ },
574
+ "Logic-pattern": {
575
+ "n": 10,
576
+ "hybrid_score": 0.0
577
+ },
578
+ "Math-arithmetic": {
579
+ "n": 33,
580
+ "hybrid_score": 0.0
581
+ },
582
+ "Math-pattern": {
583
+ "n": 14,
584
+ "hybrid_score": 0.0
585
+ },
586
+ "Math-reasoning": {
587
+ "n": 15,
588
+ "hybrid_score": 0.0247
589
+ },
590
+ "Pattern-matching": {
591
+ "n": 10,
592
+ "hybrid_score": 0.0
593
+ }
594
+ }
595
+ },
596
+ "bench-mid-6-2026": {
597
+ "score": 0.5301,
598
+ "n": 143,
599
+ "acc": 0.4266,
600
+ "acc_norm": 0.5245,
601
+ "soft_score": 0.4357,
602
+ "soft_score_norm": 0.5301,
603
+ "stderr": 0.0415,
604
+ "categories": {
605
+ "Commonsense-causality": {
606
+ "n": 5,
607
+ "acc": 0.8,
608
+ "acc_norm": 0.6
609
+ },
610
+ "Commonsense-reasoning": {
611
+ "n": 10,
612
+ "acc": 0.5,
613
+ "acc_norm": 0.4
614
+ },
615
+ "Commonsense-simulation": {
616
+ "n": 10,
617
+ "acc": 0.3,
618
+ "acc_norm": 0.4
619
+ },
620
+ "Knowledge-basic": {
621
+ "n": 7,
622
+ "acc": 0.5714,
623
+ "acc_norm": 0.8571
624
+ },
625
+ "Knowledge-definitions": {
626
+ "n": 10,
627
+ "acc": 0.1,
628
+ "acc_norm": 0.7
629
+ },
630
+ "Language-comprehension": {
631
+ "n": 10,
632
+ "acc": 0.5,
633
+ "acc_norm": 0.8
634
+ },
635
+ "Language-structure": {
636
+ "n": 10,
637
+ "acc": 0.1,
638
+ "acc_norm": 0.3
639
+ },
640
+ "Language-transformation": {
641
+ "n": 10,
642
+ "acc": 0.2,
643
+ "acc_norm": 0.4
644
+ },
645
+ "Logic-consistency": {
646
+ "n": 5,
647
+ "acc": 0.0,
648
+ "acc_norm": 0.0
649
+ },
650
+ "Logic-deduction": {
651
+ "n": 10,
652
+ "acc": 0.6,
653
+ "acc_norm": 0.7
654
+ },
655
+ "Logic-pattern": {
656
+ "n": 10,
657
+ "acc": 0.3,
658
+ "acc_norm": 0.4
659
+ },
660
+ "Math-arithmetic": {
661
+ "n": 8,
662
+ "acc": 0.875,
663
+ "acc_norm": 0.875
664
+ },
665
+ "Math-pattern": {
666
+ "n": 7,
667
+ "acc": 0.8571,
668
+ "acc_norm": 0.8571
669
+ },
670
+ "Math-reasoning": {
671
+ "n": 10,
672
+ "acc": 0.4,
673
+ "acc_norm": 0.3
674
+ },
675
+ "Pattern-generation": {
676
+ "n": 4,
677
+ "acc": 0.75,
678
+ "acc_norm": 0.5
679
+ },
680
+ "Pattern-matching": {
681
+ "n": 8,
682
+ "acc": 0.5,
683
+ "acc_norm": 0.5
684
+ },
685
+ "Pattern-recognition": {
686
+ "n": 9,
687
+ "acc": 0.3333,
688
+ "acc_norm": 0.3333
689
+ }
690
+ }
691
+ },
692
+ "bench-AGI": {
693
+ "score": null,
694
+ "n": null,
695
+ "notes": "Not yet evaluated on this tier."
696
+ }
697
+ }
698
+ },
699
+ {
700
+ "id": "liquidai-lfm2.5-350m",
701
+ "name": "LiquidAI/LFM2.5-350M",
702
+ "org": "LiquidAI",
703
+ "params_b": null,
704
+ "license": null,
705
+ "architecture": null,
706
+ "url": "https://huggingface.co/LiquidAI/LFM2.5-350M",
707
+ "model_revision": "b9d6e4e2d75f440b12a2b4d731c808004ecbbd89",
708
+ "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
709
+ "runs": {
710
+ "bench-effortless-6-2026": {
711
+ "score": 0.1375,
712
+ "n": 240,
713
+ "notes": "Exact-match, normalized.",
714
+ "categories": {
715
+ "Commonsense-reasoning": {
716
+ "n": 37,
717
+ "exact_match": 0.027
718
+ },
719
+ "Knowledge-basic": {
720
+ "n": 42,
721
+ "exact_match": 0.0952
722
+ },
723
+ "Language-comprehension": {
724
+ "n": 42,
725
+ "exact_match": 0.2143
726
+ },
727
+ "Logic-deduction": {
728
+ "n": 42,
729
+ "exact_match": 0.4286
730
+ },
731
+ "Math-arithmetic": {
732
+ "n": 40,
733
+ "exact_match": 0.0
734
+ },
735
+ "Pattern-recognition": {
736
+ "n": 37,
737
+ "exact_match": 0.027
738
+ }
739
+ }
740
+ },
741
+ "bench-easy-6-2026": {
742
+ "score": 0.358,
743
+ "n": 238,
744
+ "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
745
+ "categories": {
746
+ "Commonsense-causality": {
747
+ "n": 10,
748
+ "hybrid_score": 0.721
749
+ },
750
+ "Commonsense-reasoning": {
751
+ "n": 10,
752
+ "hybrid_score": 0.7313
753
+ },
754
+ "Commonsense-simulation": {
755
+ "n": 10,
756
+ "hybrid_score": 0.75
757
+ },
758
+ "Knowledge-basic": {
759
+ "n": 33,
760
+ "hybrid_score": 0.2424
761
+ },
762
+ "Knowledge-definitions": {
763
+ "n": 38,
764
+ "hybrid_score": 0.7996
765
+ },
766
+ "Language-comprehension": {
767
+ "n": 10,
768
+ "hybrid_score": 0.741
769
+ },
770
+ "Language-structure": {
771
+ "n": 10,
772
+ "hybrid_score": 0.3121
773
+ },
774
+ "Language-transformation": {
775
+ "n": 10,
776
+ "hybrid_score": 0.7689
777
+ },
778
+ "Logic-consistency": {
779
+ "n": 10,
780
+ "hybrid_score": 0.1
781
+ },
782
+ "Logic-deduction": {
783
+ "n": 15,
784
+ "hybrid_score": 0.2
785
+ },
786
+ "Logic-pattern": {
787
+ "n": 10,
788
+ "hybrid_score": 0.0
789
+ },
790
+ "Math-arithmetic": {
791
+ "n": 33,
792
+ "hybrid_score": 0.0
793
+ },
794
+ "Math-pattern": {
795
+ "n": 14,
796
+ "hybrid_score": 0.0
797
+ },
798
+ "Math-reasoning": {
799
+ "n": 15,
800
+ "hybrid_score": 0.1049
801
+ },
802
+ "Pattern-matching": {
803
+ "n": 10,
804
+ "hybrid_score": 0.1
805
+ }
806
+ }
807
+ },
808
+ "bench-mid-6-2026": {
809
+ "score": 0.5105,
810
+ "n": 143,
811
+ "acc": 0.4755,
812
+ "acc_norm": 0.5035,
813
+ "soft_score": 0.4825,
814
+ "soft_score_norm": 0.5105,
815
+ "stderr": 0.0417,
816
+ "categories": {
817
+ "Commonsense-causality": {
818
+ "n": 5,
819
+ "acc": 0.8,
820
+ "acc_norm": 0.8
821
+ },
822
+ "Commonsense-reasoning": {
823
+ "n": 10,
824
+ "acc": 0.5,
825
+ "acc_norm": 0.6
826
+ },
827
+ "Commonsense-simulation": {
828
+ "n": 10,
829
+ "acc": 0.4,
830
+ "acc_norm": 0.5
831
+ },
832
+ "Knowledge-basic": {
833
+ "n": 7,
834
+ "acc": 0.8571,
835
+ "acc_norm": 0.8571
836
+ },
837
+ "Knowledge-definitions": {
838
+ "n": 10,
839
+ "acc": 0.6,
840
+ "acc_norm": 0.8
841
+ },
842
+ "Language-comprehension": {
843
+ "n": 10,
844
+ "acc": 0.4,
845
+ "acc_norm": 0.8
846
+ },
847
+ "Language-structure": {
848
+ "n": 10,
849
+ "acc": 0.2,
850
+ "acc_norm": 0.1
851
+ },
852
+ "Language-transformation": {
853
+ "n": 10,
854
+ "acc": 0.2,
855
+ "acc_norm": 0.4
856
+ },
857
+ "Logic-consistency": {
858
+ "n": 5,
859
+ "acc": 0.0,
860
+ "acc_norm": 0.0
861
+ },
862
+ "Logic-deduction": {
863
+ "n": 10,
864
+ "acc": 0.3,
865
+ "acc_norm": 0.4
866
+ },
867
+ "Logic-pattern": {
868
+ "n": 10,
869
+ "acc": 0.4,
870
+ "acc_norm": 0.3
871
+ },
872
+ "Math-arithmetic": {
873
+ "n": 8,
874
+ "acc": 0.875,
875
+ "acc_norm": 0.875
876
+ },
877
+ "Math-pattern": {
878
+ "n": 7,
879
+ "acc": 0.7143,
880
+ "acc_norm": 0.7143
881
+ },
882
+ "Math-reasoning": {
883
+ "n": 10,
884
+ "acc": 0.2,
885
+ "acc_norm": 0.1
886
+ },
887
+ "Pattern-generation": {
888
+ "n": 4,
889
+ "acc": 0.75,
890
+ "acc_norm": 0.5
891
+ },
892
+ "Pattern-matching": {
893
+ "n": 8,
894
+ "acc": 0.75,
895
+ "acc_norm": 0.375
896
+ },
897
+ "Pattern-recognition": {
898
+ "n": 9,
899
+ "acc": 0.5556,
900
+ "acc_norm": 0.5556
901
+ }
902
+ }
903
+ },
904
+ "bench-AGI": {
905
+ "score": null,
906
+ "n": null,
907
+ "notes": "Not yet evaluated on this tier."
908
+ }
909
+ }
910
+ },
911
+ {
912
+ "id": "liquidai-lfm2.5-350m-base",
913
+ "name": "LiquidAI/LFM2.5-350M-Base",
914
+ "org": "LiquidAI",
915
+ "params_b": null,
916
+ "license": null,
917
+ "architecture": null,
918
+ "url": "https://huggingface.co/LiquidAI/LFM2.5-350M-Base",
919
+ "model_revision": "9960764e30892e01f29a6dc23df2533fcd8bd5ae",
920
+ "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
921
+ "runs": {
922
+ "bench-effortless-6-2026": {
923
+ "score": 0.0,
924
+ "n": 240,
925
+ "notes": "Exact-match, normalized.",
926
+ "categories": {
927
+ "Commonsense-reasoning": {
928
+ "n": 37,
929
+ "exact_match": 0.0
930
+ },
931
+ "Knowledge-basic": {
932
+ "n": 42,
933
+ "exact_match": 0.0
934
+ },
935
+ "Language-comprehension": {
936
+ "n": 42,
937
+ "exact_match": 0.0
938
+ },
939
+ "Logic-deduction": {
940
+ "n": 42,
941
+ "exact_match": 0.0
942
+ },
943
+ "Math-arithmetic": {
944
+ "n": 40,
945
+ "exact_match": 0.0
946
+ },
947
+ "Pattern-recognition": {
948
+ "n": 37,
949
+ "exact_match": 0.0
950
+ }
951
+ }
952
+ },
953
+ "bench-easy-6-2026": {
954
+ "score": 0.2788,
955
+ "n": 238,
956
+ "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
957
+ "categories": {
958
+ "Commonsense-causality": {
959
+ "n": 10,
960
+ "hybrid_score": 0.7048
961
+ },
962
+ "Commonsense-reasoning": {
963
+ "n": 10,
964
+ "hybrid_score": 0.6998
965
+ },
966
+ "Commonsense-simulation": {
967
+ "n": 10,
968
+ "hybrid_score": 0.744
969
+ },
970
+ "Knowledge-basic": {
971
+ "n": 33,
972
+ "hybrid_score": 0.0
973
+ },
974
+ "Knowledge-definitions": {
975
+ "n": 38,
976
+ "hybrid_score": 0.7958
977
+ },
978
+ "Language-comprehension": {
979
+ "n": 10,
980
+ "hybrid_score": 0.7201
981
+ },
982
+ "Language-structure": {
983
+ "n": 10,
984
+ "hybrid_score": 0.146
985
+ },
986
+ "Language-transformation": {
987
+ "n": 10,
988
+ "hybrid_score": 0.5558
989
+ },
990
+ "Logic-consistency": {
991
+ "n": 10,
992
+ "hybrid_score": 0.0
993
+ },
994
+ "Logic-deduction": {
995
+ "n": 15,
996
+ "hybrid_score": 0.0
997
+ },
998
+ "Logic-pattern": {
999
+ "n": 10,
1000
+ "hybrid_score": 0.0
1001
+ },
1002
+ "Math-arithmetic": {
1003
+ "n": 33,
1004
+ "hybrid_score": 0.0
1005
+ },
1006
+ "Math-pattern": {
1007
+ "n": 14,
1008
+ "hybrid_score": 0.0
1009
+ },
1010
+ "Math-reasoning": {
1011
+ "n": 15,
1012
+ "hybrid_score": 0.0271
1013
+ },
1014
+ "Pattern-matching": {
1015
+ "n": 10,
1016
+ "hybrid_score": 0.0
1017
+ }
1018
+ }
1019
+ },
1020
+ "bench-mid-6-2026": {
1021
+ "score": 0.5105,
1022
+ "n": 143,
1023
+ "acc": 0.3986,
1024
+ "acc_norm": 0.5035,
1025
+ "soft_score": 0.4091,
1026
+ "soft_score_norm": 0.5105,
1027
+ "stderr": 0.0417,
1028
+ "categories": {
1029
+ "Commonsense-causality": {
1030
+ "n": 5,
1031
+ "acc": 0.8,
1032
+ "acc_norm": 0.6
1033
+ },
1034
+ "Commonsense-reasoning": {
1035
+ "n": 10,
1036
+ "acc": 0.4,
1037
+ "acc_norm": 0.5
1038
+ },
1039
+ "Commonsense-simulation": {
1040
+ "n": 10,
1041
+ "acc": 0.3,
1042
+ "acc_norm": 0.3
1043
+ },
1044
+ "Knowledge-basic": {
1045
+ "n": 7,
1046
+ "acc": 0.4286,
1047
+ "acc_norm": 0.5714
1048
+ },
1049
+ "Knowledge-definitions": {
1050
+ "n": 10,
1051
+ "acc": 0.1,
1052
+ "acc_norm": 0.7
1053
+ },
1054
+ "Language-comprehension": {
1055
+ "n": 10,
1056
+ "acc": 0.3,
1057
+ "acc_norm": 0.8
1058
+ },
1059
+ "Language-structure": {
1060
+ "n": 10,
1061
+ "acc": 0.0,
1062
+ "acc_norm": 0.2
1063
+ },
1064
+ "Language-transformation": {
1065
+ "n": 10,
1066
+ "acc": 0.5,
1067
+ "acc_norm": 0.7
1068
+ },
1069
+ "Logic-consistency": {
1070
+ "n": 5,
1071
+ "acc": 0.0,
1072
+ "acc_norm": 0.0
1073
+ },
1074
+ "Logic-deduction": {
1075
+ "n": 10,
1076
+ "acc": 0.4,
1077
+ "acc_norm": 0.6
1078
+ },
1079
+ "Logic-pattern": {
1080
+ "n": 10,
1081
+ "acc": 0.4,
1082
+ "acc_norm": 0.5
1083
+ },
1084
+ "Math-arithmetic": {
1085
+ "n": 8,
1086
+ "acc": 0.875,
1087
+ "acc_norm": 0.875
1088
+ },
1089
+ "Math-pattern": {
1090
+ "n": 7,
1091
+ "acc": 0.7143,
1092
+ "acc_norm": 0.7143
1093
+ },
1094
+ "Math-reasoning": {
1095
+ "n": 10,
1096
+ "acc": 0.4,
1097
+ "acc_norm": 0.3
1098
+ },
1099
+ "Pattern-generation": {
1100
+ "n": 4,
1101
+ "acc": 0.75,
1102
+ "acc_norm": 0.5
1103
+ },
1104
+ "Pattern-matching": {
1105
+ "n": 8,
1106
+ "acc": 0.625,
1107
+ "acc_norm": 0.375
1108
+ },
1109
+ "Pattern-recognition": {
1110
+ "n": 9,
1111
+ "acc": 0.2222,
1112
+ "acc_norm": 0.2222
1113
+ }
1114
+ }
1115
+ },
1116
+ "bench-AGI": {
1117
+ "score": null,
1118
+ "n": null,
1119
+ "notes": "Not yet evaluated on this tier."
1120
+ }
1121
+ }
1122
+ },
1123
+ {
1124
  "id": "liquidai-lfm2.5-230m",
1125
  "name": "LiquidAI/LFM2.5-230M",
1126
  "org": "LiquidAI",