NEXORA / nexora /evaluation.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
1.34 kB
"""Metrics without executing untrusted benchmark code on the host."""
import math
import statistics
def pass_at_k(n, correct, k):
if not 0 <= correct <= n or not 1 <= k <= n:
raise ValueError("Require 0<=correct<=n and 1<=k<=n")
return 1.0 if n-correct < k else 1-math.comb(n-correct, k)/math.comb(n, k)
def wilson(successes, n, z=1.96):
if n < 1 or not 0 <= successes <= n:
raise ValueError("Invalid counts")
p = successes/n
center = (p+z*z/(2*n))/(1+z*z/n)
width = z*math.sqrt(p*(1-p)/n+z*z/(4*n*n))/(1+z*z/n)
return [max(0, center-width), min(1, center+width)]
def percentiles(values):
if not values or not all(math.isfinite(x) and x >= 0 for x in values):
raise ValueError("Expected finite nonnegative measurements")
ordered = sorted(values)
return {f"p{p}": ordered[max(0, math.ceil(len(ordered)*p/100)-1)] for p in (50, 95, 99)}
def word_error_rate(reference, hypothesis):
a, b = reference.split(), hypothesis.split()
if not a:
raise ValueError("Reference must contain words")
row = list(range(len(b)+1))
for i, word in enumerate(a, 1):
next_row = [i]
for j, other in enumerate(b, 1):
next_row.append(min(next_row[-1]+1, row[j]+1, row[j-1]+(word != other)))
row = next_row
return row[-1]/len(a)