ereniko commited on
Commit
60ae06f
·
verified ·
1 Parent(s): 7b48a90

Add conversate v2 base

Browse files
Files changed (1) hide show
  1. models.json +396 -0
models.json CHANGED
@@ -4164,6 +4164,402 @@
4164
  "notes": "Not yet evaluated on this tier."
4165
  }
4166
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4167
  }
4168
  ],
4169
  "t2i_benchmarks": [
 
4164
  "notes": "Not yet evaluated on this tier."
4165
  }
4166
  }
4167
+ },
4168
+ {
4169
+ "id": "ivmelabs-ivme-conversate-v2-base",
4170
+ "name": "IvmeLabs/Ivme-Conversate-v2-Base",
4171
+ "org": "IvmeLabs",
4172
+ "params_b": null,
4173
+ "license": null,
4174
+ "architecture": null,
4175
+ "url": "https://huggingface.co/IvmeLabs/Ivme-Conversate-v2-Base",
4176
+ "model_revision": "8b157ab2f11c8fe1897b092b8ecd946dbe65a870",
4177
+ "script_sha256": "955411f9971c4c26a6eeb3aa43b1694fe66ac430a139011faa915543746d2921",
4178
+ "runs": {
4179
+ "bench-effortless-7-2026": {
4180
+ "score": 0.0033,
4181
+ "n": 300,
4182
+ "stderr": 0.0033,
4183
+ "metrics": {
4184
+ "generative": {
4185
+ "exact_match": 0.0033,
4186
+ "hybrid_score": 0.2322
4187
+ },
4188
+ "loglikelihood": {
4189
+ "acc": 0.4233,
4190
+ "acc_norm": 0.4,
4191
+ "soft_score": 0.4233,
4192
+ "soft_score_norm": 0.4
4193
+ }
4194
+ },
4195
+ "categories": {
4196
+ "Commonsense-causality": {
4197
+ "n": 18,
4198
+ "score": 0.0,
4199
+ "exact_match": 0.0,
4200
+ "acc_norm": 0.3889
4201
+ },
4202
+ "Commonsense-reasoning": {
4203
+ "n": 18,
4204
+ "score": 0.0,
4205
+ "exact_match": 0.0,
4206
+ "acc_norm": 0.5556
4207
+ },
4208
+ "Commonsense-simulation": {
4209
+ "n": 18,
4210
+ "score": 0.0,
4211
+ "exact_match": 0.0,
4212
+ "acc_norm": 0.3889
4213
+ },
4214
+ "Knowledge-basic": {
4215
+ "n": 18,
4216
+ "score": 0.0,
4217
+ "exact_match": 0.0,
4218
+ "acc_norm": 0.2778
4219
+ },
4220
+ "Knowledge-definitions": {
4221
+ "n": 18,
4222
+ "score": 0.0,
4223
+ "exact_match": 0.0,
4224
+ "acc_norm": 0.5556
4225
+ },
4226
+ "Language-comprehension": {
4227
+ "n": 18,
4228
+ "score": 0.0,
4229
+ "exact_match": 0.0,
4230
+ "acc_norm": 0.7222
4231
+ },
4232
+ "Language-structure": {
4233
+ "n": 18,
4234
+ "score": 0.0,
4235
+ "exact_match": 0.0,
4236
+ "acc_norm": 0.2222
4237
+ },
4238
+ "Language-transformation": {
4239
+ "n": 18,
4240
+ "score": 0.0,
4241
+ "exact_match": 0.0,
4242
+ "acc_norm": 0.2778
4243
+ },
4244
+ "Logic-consistency": {
4245
+ "n": 18,
4246
+ "score": 0.0,
4247
+ "exact_match": 0.0,
4248
+ "acc_norm": 0.0556
4249
+ },
4250
+ "Logic-deduction": {
4251
+ "n": 18,
4252
+ "score": 0.0,
4253
+ "exact_match": 0.0,
4254
+ "acc_norm": 0.7222
4255
+ },
4256
+ "Logic-pattern": {
4257
+ "n": 18,
4258
+ "score": 0.0556,
4259
+ "exact_match": 0.0556,
4260
+ "acc_norm": 0.8889
4261
+ },
4262
+ "Math-arithmetic": {
4263
+ "n": 17,
4264
+ "score": 0.0,
4265
+ "exact_match": 0.0,
4266
+ "acc_norm": 0.4118
4267
+ },
4268
+ "Math-pattern": {
4269
+ "n": 17,
4270
+ "score": 0.0,
4271
+ "exact_match": 0.0,
4272
+ "acc_norm": 0.1765
4273
+ },
4274
+ "Math-reasoning": {
4275
+ "n": 17,
4276
+ "score": 0.0,
4277
+ "exact_match": 0.0,
4278
+ "acc_norm": 0.2941
4279
+ },
4280
+ "Pattern-generation": {
4281
+ "n": 17,
4282
+ "score": 0.0,
4283
+ "exact_match": 0.0,
4284
+ "acc_norm": 0.4706
4285
+ },
4286
+ "Pattern-matching": {
4287
+ "n": 17,
4288
+ "score": 0.0,
4289
+ "exact_match": 0.0,
4290
+ "acc_norm": 0.2941
4291
+ },
4292
+ "Pattern-recognition": {
4293
+ "n": 17,
4294
+ "score": 0.0,
4295
+ "exact_match": 0.0,
4296
+ "acc_norm": 0.0588
4297
+ }
4298
+ }
4299
+ },
4300
+ "bench-easy-7-2026": {
4301
+ "score": 0.2391,
4302
+ "n": 300,
4303
+ "stderr": 0.0185,
4304
+ "metrics": {
4305
+ "generative": {
4306
+ "exact_match": 0.0,
4307
+ "hybrid_score": 0.2391
4308
+ },
4309
+ "loglikelihood": {
4310
+ "acc": 0.33,
4311
+ "acc_norm": 0.37,
4312
+ "soft_score": 0.33,
4313
+ "soft_score_norm": 0.37
4314
+ }
4315
+ },
4316
+ "categories": {
4317
+ "Commonsense-causality": {
4318
+ "n": 18,
4319
+ "score": 0.6716,
4320
+ "exact_match": 0.0,
4321
+ "acc_norm": 0.5
4322
+ },
4323
+ "Commonsense-reasoning": {
4324
+ "n": 18,
4325
+ "score": 0.6388,
4326
+ "exact_match": 0.0,
4327
+ "acc_norm": 0.5
4328
+ },
4329
+ "Commonsense-simulation": {
4330
+ "n": 18,
4331
+ "score": 0.6547,
4332
+ "exact_match": 0.0,
4333
+ "acc_norm": 0.6111
4334
+ },
4335
+ "Knowledge-basic": {
4336
+ "n": 18,
4337
+ "score": 0.0,
4338
+ "exact_match": 0.0,
4339
+ "acc_norm": 0.2222
4340
+ },
4341
+ "Knowledge-definitions": {
4342
+ "n": 18,
4343
+ "score": 0.697,
4344
+ "exact_match": 0.0,
4345
+ "acc_norm": 0.5
4346
+ },
4347
+ "Language-comprehension": {
4348
+ "n": 18,
4349
+ "score": 0.6766,
4350
+ "exact_match": 0.0,
4351
+ "acc_norm": 0.7222
4352
+ },
4353
+ "Language-structure": {
4354
+ "n": 18,
4355
+ "score": 0.1402,
4356
+ "exact_match": 0.0,
4357
+ "acc_norm": 0.2222
4358
+ },
4359
+ "Language-transformation": {
4360
+ "n": 18,
4361
+ "score": 0.5056,
4362
+ "exact_match": 0.0,
4363
+ "acc_norm": 0.4444
4364
+ },
4365
+ "Logic-consistency": {
4366
+ "n": 18,
4367
+ "score": 0.0,
4368
+ "exact_match": 0.0,
4369
+ "acc_norm": 0.0
4370
+ },
4371
+ "Logic-deduction": {
4372
+ "n": 18,
4373
+ "score": 0.0,
4374
+ "exact_match": 0.0,
4375
+ "acc_norm": 0.6667
4376
+ },
4377
+ "Logic-pattern": {
4378
+ "n": 18,
4379
+ "score": 0.0,
4380
+ "exact_match": 0.0,
4381
+ "acc_norm": 0.6111
4382
+ },
4383
+ "Math-arithmetic": {
4384
+ "n": 17,
4385
+ "score": 0.0,
4386
+ "exact_match": 0.0,
4387
+ "acc_norm": 0.1176
4388
+ },
4389
+ "Math-pattern": {
4390
+ "n": 17,
4391
+ "score": 0.0,
4392
+ "exact_match": 0.0,
4393
+ "acc_norm": 0.1176
4394
+ },
4395
+ "Math-reasoning": {
4396
+ "n": 17,
4397
+ "score": 0.0,
4398
+ "exact_match": 0.0,
4399
+ "acc_norm": 0.3529
4400
+ },
4401
+ "Pattern-generation": {
4402
+ "n": 17,
4403
+ "score": 0.0,
4404
+ "exact_match": 0.0,
4405
+ "acc_norm": 0.3529
4406
+ },
4407
+ "Pattern-matching": {
4408
+ "n": 17,
4409
+ "score": 0.0,
4410
+ "exact_match": 0.0,
4411
+ "acc_norm": 0.2353
4412
+ },
4413
+ "Pattern-recognition": {
4414
+ "n": 17,
4415
+ "score": 0.0,
4416
+ "exact_match": 0.0,
4417
+ "acc_norm": 0.0588
4418
+ }
4419
+ }
4420
+ },
4421
+ "bench-mid-7-2026": {
4422
+ "score": 0.3133,
4423
+ "n": 300,
4424
+ "stderr": 0.0268,
4425
+ "metrics": {
4426
+ "generative": {
4427
+ "exact_match": 0.0,
4428
+ "hybrid_score": 0.225
4429
+ },
4430
+ "loglikelihood": {
4431
+ "acc": 0.2633,
4432
+ "acc_norm": 0.3133,
4433
+ "soft_score": 0.2633,
4434
+ "soft_score_norm": 0.3133
4435
+ }
4436
+ },
4437
+ "categories": {
4438
+ "Commonsense-causality": {
4439
+ "n": 18,
4440
+ "score": 0.5,
4441
+ "exact_match": 0.0,
4442
+ "acc_norm": 0.5
4443
+ },
4444
+ "Commonsense-reasoning": {
4445
+ "n": 18,
4446
+ "score": 0.3889,
4447
+ "exact_match": 0.0,
4448
+ "acc_norm": 0.3889
4449
+ },
4450
+ "Commonsense-simulation": {
4451
+ "n": 18,
4452
+ "score": 0.2778,
4453
+ "exact_match": 0.0,
4454
+ "acc_norm": 0.2778
4455
+ },
4456
+ "Knowledge-basic": {
4457
+ "n": 18,
4458
+ "score": 0.3889,
4459
+ "exact_match": 0.0,
4460
+ "acc_norm": 0.3889
4461
+ },
4462
+ "Knowledge-definitions": {
4463
+ "n": 18,
4464
+ "score": 0.6111,
4465
+ "exact_match": 0.0,
4466
+ "acc_norm": 0.6111
4467
+ },
4468
+ "Language-comprehension": {
4469
+ "n": 18,
4470
+ "score": 0.3889,
4471
+ "exact_match": 0.0,
4472
+ "acc_norm": 0.3889
4473
+ },
4474
+ "Language-structure": {
4475
+ "n": 18,
4476
+ "score": 0.3889,
4477
+ "exact_match": 0.0,
4478
+ "acc_norm": 0.3889
4479
+ },
4480
+ "Language-transformation": {
4481
+ "n": 18,
4482
+ "score": 0.3333,
4483
+ "exact_match": 0.0,
4484
+ "acc_norm": 0.3333
4485
+ },
4486
+ "Logic-consistency": {
4487
+ "n": 18,
4488
+ "score": 0.0,
4489
+ "exact_match": 0.0,
4490
+ "acc_norm": 0.0
4491
+ },
4492
+ "Logic-deduction": {
4493
+ "n": 18,
4494
+ "score": 0.3889,
4495
+ "exact_match": 0.0,
4496
+ "acc_norm": 0.3889
4497
+ },
4498
+ "Logic-pattern": {
4499
+ "n": 18,
4500
+ "score": 0.3333,
4501
+ "exact_match": 0.0,
4502
+ "acc_norm": 0.3333
4503
+ },
4504
+ "Math-arithmetic": {
4505
+ "n": 17,
4506
+ "score": 0.1176,
4507
+ "exact_match": 0.0,
4508
+ "acc_norm": 0.1176
4509
+ },
4510
+ "Math-pattern": {
4511
+ "n": 17,
4512
+ "score": 0.0588,
4513
+ "exact_match": 0.0,
4514
+ "acc_norm": 0.0588
4515
+ },
4516
+ "Math-reasoning": {
4517
+ "n": 17,
4518
+ "score": 0.1765,
4519
+ "exact_match": 0.0,
4520
+ "acc_norm": 0.1765
4521
+ },
4522
+ "Pattern-generation": {
4523
+ "n": 17,
4524
+ "score": 0.2941,
4525
+ "exact_match": 0.0,
4526
+ "acc_norm": 0.2941
4527
+ },
4528
+ "Pattern-matching": {
4529
+ "n": 17,
4530
+ "score": 0.5882,
4531
+ "exact_match": 0.0,
4532
+ "acc_norm": 0.5882
4533
+ },
4534
+ "Pattern-recognition": {
4535
+ "n": 17,
4536
+ "score": 0.0588,
4537
+ "exact_match": 0.0,
4538
+ "acc_norm": 0.0588
4539
+ }
4540
+ }
4541
+ },
4542
+ "bench-effortless-6-2026": {
4543
+ "score": null,
4544
+ "n": null,
4545
+ "notes": "Not yet evaluated on this tier."
4546
+ },
4547
+ "bench-easy-6-2026": {
4548
+ "score": null,
4549
+ "n": null,
4550
+ "notes": "Not yet evaluated on this tier."
4551
+ },
4552
+ "bench-mid-6-2026": {
4553
+ "score": null,
4554
+ "n": null,
4555
+ "notes": "Not yet evaluated on this tier."
4556
+ },
4557
+ "bench-AGI": {
4558
+ "score": null,
4559
+ "n": null,
4560
+ "notes": "Not yet evaluated on this tier."
4561
+ }
4562
+ }
4563
  }
4564
  ],
4565
  "t2i_benchmarks": [