Download nexora/evaluation.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 1.34 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/evaluation.py
- Command line
-
hf download hf://devildasdf/NEXORA/nexora/evaluation.py
-
curl -L -o evaluation.py https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/evaluation.py
1.34 kB
| """Metrics without executing untrusted benchmark code on the host.""" | |
| import math | |
| import statistics | |
| def pass_at_k(n, correct, k): | |
| if not 0 <= correct <= n or not 1 <= k <= n: | |
| raise ValueError("Require 0<=correct<=n and 1<=k<=n") | |
| return 1.0 if n-correct < k else 1-math.comb(n-correct, k)/math.comb(n, k) | |
| def wilson(successes, n, z=1.96): | |
| if n < 1 or not 0 <= successes <= n: | |
| raise ValueError("Invalid counts") | |
| p = successes/n | |
| center = (p+z*z/(2*n))/(1+z*z/n) | |
| width = z*math.sqrt(p*(1-p)/n+z*z/(4*n*n))/(1+z*z/n) | |
| return [max(0, center-width), min(1, center+width)] | |
| def percentiles(values): | |
| if not values or not all(math.isfinite(x) and x >= 0 for x in values): | |
| raise ValueError("Expected finite nonnegative measurements") | |
| ordered = sorted(values) | |
| return {f"p{p}": ordered[max(0, math.ceil(len(ordered)*p/100)-1)] for p in (50, 95, 99)} | |
| def word_error_rate(reference, hypothesis): | |
| a, b = reference.split(), hypothesis.split() | |
| if not a: | |
| raise ValueError("Reference must contain words") | |
| row = list(range(len(b)+1)) | |
| for i, word in enumerate(a, 1): | |
| next_row = [i] | |
| for j, other in enumerate(b, 1): | |
| next_row.append(min(next_row[-1]+1, row[j]+1, row[j-1]+(word != other))) | |
| row = next_row | |
| return row[-1]/len(a) | |