Spaces:
Sleeping
Sleeping
File size: 9,104 Bytes
0865492 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 | /**
* List of supported Evaluation Frameworks supported in the `eval.yaml` file in benchmarks datasets.
*/
export declare const EVALUATION_FRAMEWORKS: {
readonly exgentic: {
readonly name: "exgentic";
readonly description: "Exgentic is an open evaluation framework for general-purpose AI agents across diverse domains and benchmarks.";
readonly url: "https://github.com/Exgentic/exgentic";
};
readonly "inspect-ai": {
readonly name: "inspect-ai";
readonly description: "Inspect AI is an open-source framework for large language model evaluations.";
readonly url: "https://inspect.aisi.org.uk/";
};
readonly "math-arena": {
readonly name: "math-arena";
readonly description: "MathArena is a platform for evaluation of LLMs on latest math competitions and olympiads.";
readonly url: "https://github.com/eth-sri/matharena";
};
readonly mteb: {
readonly name: "mteb";
readonly description: "Multimodal toolbox for evaluating embeddings and retrieval systems.";
readonly url: "https://github.com/embeddings-benchmark/mteb";
};
readonly "olmocr-bench": {
readonly name: "olmocr-bench";
readonly description: "olmOCR-Bench is a framework for evaluating document-level OCR of various tools.";
readonly url: "https://github.com/allenai/olmocr/tree/main/olmocr/bench";
};
readonly harbor: {
readonly name: "harbor";
readonly description: "Harbor is a framework for evaluating and optimizing agents and language models.";
readonly url: "https://github.com/laude-institute/harbor";
};
readonly ifstruct: {
readonly name: "ifstruct";
readonly description: "IFStruct is a benchmark for structured-output compliance: whether a model produces valid JSON/YAML that follows a requested schema, scored without constrained decoding.";
readonly url: "https://github.com/Liquid4All/ifstruct";
};
readonly pier: {
readonly name: "pier";
readonly description: "Pier is a Harbor fork built for DeepSWE, with stronger support for CLI agents in no-internet tasks and more faithful, consistent agent trajectories.";
readonly url: "https://github.com/datacurve-ai/pier";
};
readonly "redline-bench": {
readonly name: "redline-bench";
readonly description: "RedlineBench measures multi-turn contract redlining: agents produce tracked-change .docx edits that are graded against attorney-authored weighted rubrics by an LLM judge panel across five dimensions. Report: https://intelligence.crosby.ai/benchmark/";
readonly url: "https://github.com/crosbylegal/redline-bench";
};
readonly archipelago: {
readonly name: "archipelago";
readonly description: "Archipelago is a system for running and evaluating AI agents against MCP applications.";
readonly url: "https://github.com/Mercor-Intelligence/archipelago";
};
readonly benchflow: {
readonly name: "benchflow";
readonly description: "BenchFlow is an evaluation framework for AI agents on professional, skill-aware workflows. It powers SkillsBench and runs containerized agent trials with paired with-skills / without-skills configurations.";
readonly url: "https://github.com/benchflow-ai/benchflow";
};
readonly "apex-evals": {
readonly name: "apex-evals";
readonly description: "APEX Evals is a benchmark suite and evaluation harness for evaluating large language models.";
readonly url: "https://github.com/Mercor-Intelligence/apex-evals";
};
readonly "screenspot-pro": {
readonly name: "screenspot-pro";
readonly description: "ScreenSpot-Pro is a GUI grounding benchmark designed to evaluate how well AI agents can locate and identify UI elements across professional software applications in high-resolution screenshots, covering 1,585 annotated images from 26 professional tools.";
readonly url: "https://github.com/likaixin2000/ScreenSpot-Pro-GUI-Grounding";
};
readonly "swe-bench": {
readonly name: "swe-bench";
readonly description: "SWE Bench is a framework for evaluating the performance of LLMs on software engineering tasks.";
readonly url: "https://github.com/swe-bench/swe-bench";
};
readonly "swe-bench-pro": {
readonly name: "swe-bench-pro";
readonly description: "SWE-Bench Pro is a challenging benchmark evaluating LLMs/Agents on long-horizon software engineering tasks.";
readonly url: "https://github.com/scaleapi/SWE-bench_Pro-os";
};
readonly "nemo-evaluator": {
readonly name: "nemo-evaluator";
readonly description: "NeMo Evaluator is an open-source platform for robust, reproducible, and scalable evaluation of Large Language Models across 100+ benchmarks.";
readonly url: "https://github.com/NVIDIA-NeMo/Evaluator";
};
readonly "yc-bench": {
readonly name: "yc-bench";
readonly description: "YC Bench is a long-horizon deterministic benchmark for LLM agents. The agent plays CEO of an AI startup over a simulated 1–3 year run.";
readonly url: "https://github.com/collinear-ai/yc-bench";
};
readonly "open-asr-leaderboard": {
readonly name: "open-asr-leaderboard";
readonly description: "The Open ASR Leaderboard ranks and evaluates speech recognition models.";
readonly url: "https://github.com/huggingface/open_asr_leaderboard";
};
readonly mdpbench: {
readonly name: "mdpbench";
readonly description: "MDPBench is a benchmark for evaluating multilingual document parsing across digital, photographed, Latin, and non-Latin document subsets.";
readonly url: "https://github.com/Yuliang-Liu/MultimodalOCR";
};
readonly parsebench: {
readonly name: "parsebench";
readonly description: "ParseBench is a benchmark for evaluating document parsing systems on real-world enterprise documents across tables, charts, content faithfulness, semantic formatting, and visual grounding.";
readonly url: "https://github.com/run-llama/ParseBench";
};
readonly "video-mme-v2": {
readonly name: "video-mme-v2";
readonly description: "Video-MME-v2 is a benchmark for evaluating the next stage of video understanding capabilities of multimodal large language models.";
readonly url: "https://github.com/MME-Benchmarks/Video-MME-v2";
};
readonly "claw-eval": {
readonly name: "claw-eval";
readonly description: "CLAW-Eval is an evaluation framework for assessing LLMs as autonomous agents across 300 human-verified tasks covering communication, finance, and productivity domains.";
readonly url: "https://github.com/claw-eval/claw-eval";
};
readonly researchclawbench: {
readonly name: "researchclawbench";
readonly description: "ResearchClawBench is a benchmark for evaluating AI agents on end-to-end scientific research tasks, from reading data and related work to producing code, figures, and publication-style reports.";
readonly url: "https://github.com/InternScience/ResearchClawBench";
};
readonly pbench: {
readonly name: "pbench";
readonly description: "PBench is a multi-level referring expression segmentation benchmark for evaluating vision-language perception across a structured hierarchy of skills.";
readonly url: "https://github.com/tiiuae/Falcon-Perception";
};
readonly wildclawbench: {
readonly name: "wildclawbench";
readonly description: "WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.";
readonly url: "https://github.com/InternLM/WildClawBench";
};
readonly wbench: {
readonly name: "wbench";
readonly description: "WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.";
readonly url: "https://github.com/meituan-longcat/WBench";
};
readonly nanofold: {
readonly name: "nanofold";
readonly description: "nanoFold is a data-efficiency benchmark for protein structure prediction. Its goal is to evaluate models on scenarios with scarce data.";
readonly url: "https://github.com/ChrisHayduk/nanoFold-Competition";
};
readonly mmmu: {
readonly name: "mmmu";
readonly description: "MMMU is a new benchmark designed to evaluate multimodal models on massive multi-discipline tasks demanding college-level subject knowledge and deliberate reasoning.";
readonly url: "https://mmmu-benchmark.github.io/";
};
};
//# sourceMappingURL=eval.d.ts.map |