Kalpana-Multi-Model-Benchmark / tests /test_statistics.py
MaduRox's picture
Upload folder using huggingface_hub
e3584eb verified
Raw
History Blame Contribute Delete
686 Bytes
from benchmark.statistics import run_mcnemar_test
def test_mcnemar_equal_outcomes():
plain = [True, True, False, False]
rif = [True, True, False, False]
stats = run_mcnemar_test(plain, rif)
assert stats["p_value"] == 1.0
assert stats["statistic"] == 0.0
assert stats["accuracy_difference"] == 0.0
def test_mcnemar_significant_diff():
# RIF significantly outperforming Plain
plain = [True] * 5 + [False] * 20
rif = [True] * 20 + [False] * 5
stats = run_mcnemar_test(plain, rif)
assert stats["sample_size"] == 25
assert stats["accuracy_difference"] == 0.6
assert stats["ci_lower"] <= stats["accuracy_difference"] <= stats["ci_upper"]