Spaces:

PKU-JX-LAB
/

Molecular-Dynamics-Benchmark

Sleeping

App Files Files Community

FredericFan commited on 8 days ago

Commit

7f5a423

1 Parent(s): f1e78da

updated

Browse files

Files changed (7) hide show

.gitignore +1 -0
README.md +28 -38
app.py +201 -171
requirements.txt +1 -15
src/about.py +81 -39
src/display/utils.py +36 -48
src/envs.py +6 -8

.gitignore CHANGED Viewed

@@ -11,3 +11,4 @@ eval-results/
 eval-queue-bk/
 eval-results-bk/
 logs/

 eval-queue-bk/
 eval-results-bk/
 logs/
+/paper_latex

README.md CHANGED Viewed

@@ -1,48 +1,38 @@
 ---
-title: MD Benchmark
-emoji: 🥇
-colorFrom: green
 colorTo: indigo
 sdk: gradio
 app_file: app.py
 pinned: true
 license: apache-2.0
-short_description: Molecular Dynamics eval benchmark
-sdk_version: 5.43.1
 tags:
 - leaderboard
 ---
-# Start the configuration
-Most of the variables to change for a default leaderboard are in `src/env.py` (replace the path for your leaderboard) and `src/about.py` (for tasks).
-Results files should have the following format and be stored as json files:
-```json
-{
-    "config": {
-        "model_dtype": "torch.float16", # or torch.bfloat16 or 8bit or 4bit
-        "model_name": "path of the model on the hub: org/model",
-        "model_sha": "revision on the hub",
-    },
-    "results": {
-        "task_name": {
-            "metric_name": score,
-        },
-        "task_name2": {
-            "metric_name": score,
-        }
-    }
-}
-```
-Request files are created automatically by this tool.
-If you encounter problem on the space, don't hesitate to restart it to remove the create eval-queue, eval-queue-bk, eval-results and eval-results-bk created folder.
-# Code logic for more complex edits
-You'll find
-- the main table' columns names and properties in `src/display/utils.py`
-- the logic to read all results and request files, then convert them in dataframe lines, in `src/leaderboard/read_evals.py`, and `src/populate.py`
-- the logic to allow or filter submissions in `src/submission/submit.py` and `src/submission/check_validity.py`

 ---
+title: MD-EvalBench Leaderboard
+emoji: 🧪
+colorFrom: blue
 colorTo: indigo
 sdk: gradio
+sdk_version: 5.43.1
 app_file: app.py
 pinned: true
 license: apache-2.0
+short_description: Molecular Dynamics LLM Benchmark Leaderboard
 tags:
 - leaderboard
+- molecular-dynamics
+- LAMMPS
+- AI-for-Science
 ---
+## MD-EvalBench: Molecular Dynamics LLM Benchmark
+The first comprehensive benchmark for evaluating Large Language Models in the Molecular Dynamics domain, from the paper **"MDAgent2: Large Language Model for Code Generation and Knowledge Q&A in Molecular Dynamics"**.
+## Benchmark Components
+- **MD-KnowledgeEval** (336 questions): Theoretical knowledge in molecular dynamics
+- **LAMMPS-SyntaxEval** (368 questions): LAMMPS command and syntax understanding
+- **LAMMPS-CodeGenEval** (566 tasks): Automatic LAMMPS code generation
+## Leaderboard Tabs
+1. **QA Benchmark**: Knowledge and syntax QA performance comparison
+2. **Code Generation**: LAMMPS code generation performance across methods (Direct Prompting, MDAgent, MDAgent2-RUNTIME)
+## Configuration
+- Main leaderboard data is in `app.py`
+- Task definitions in `src/about.py`
+- Column definitions in `src/display/utils.py`

app.py CHANGED Viewed

@@ -1,204 +1,234 @@
 import gradio as gr
-from gradio_leaderboard import Leaderboard, ColumnFilter, SelectColumns
 import pandas as pd
-from apscheduler.schedulers.background import BackgroundScheduler
-from huggingface_hub import snapshot_download
 from src.about import (
     CITATION_BUTTON_LABEL,
     CITATION_BUTTON_TEXT,
-    EVALUATION_QUEUE_TEXT,
     INTRODUCTION_TEXT,
     LLM_BENCHMARKS_TEXT,
     TITLE,
 )
 from src.display.css_html_js import custom_css
-from src.display.utils import (
-    BENCHMARK_COLS,
-    COLS,
-    EVAL_COLS,
-    EVAL_TYPES,
-    AutoEvalColumn,
-    ModelType,
-    fields,
-    WeightType,
-    Precision
-)
-from src.envs import API, EVAL_REQUESTS_PATH, EVAL_RESULTS_PATH, QUEUE_REPO, REPO_ID, RESULTS_REPO, TOKEN
-from src.populate import get_evaluation_queue_df, get_leaderboard_df
-from src.submission.submit import add_new_eval
-def restart_space():
-    API.restart_space(repo_id=REPO_ID)
-### Space initialisation
-try:
-    print(EVAL_REQUESTS_PATH)
-    snapshot_download(
-        repo_id=QUEUE_REPO, local_dir=EVAL_REQUESTS_PATH, repo_type="dataset", tqdm_class=None, etag_timeout=30, token=TOKEN
-    )
-except Exception:
-    restart_space()
-try:
-    print(EVAL_RESULTS_PATH)
-    snapshot_download(
-        repo_id=RESULTS_REPO, local_dir=EVAL_RESULTS_PATH, repo_type="dataset", tqdm_class=None, etag_timeout=30, token=TOKEN
-    )
-except Exception:
-    restart_space()
-LEADERBOARD_DF = get_leaderboard_df(EVAL_RESULTS_PATH, EVAL_REQUESTS_PATH, COLS, BENCHMARK_COLS)
-(
-    finished_eval_queue_df,
-    running_eval_queue_df,
-    pending_eval_queue_df,
-) = get_evaluation_queue_df(EVAL_REQUESTS_PATH, EVAL_COLS)
-def init_leaderboard(dataframe):
-    if dataframe is None or dataframe.empty:
-        raise ValueError("Leaderboard DataFrame is empty or None.")
-    return Leaderboard(
-        value=dataframe,
-        datatype=[c.type for c in fields(AutoEvalColumn)],
-        select_columns=SelectColumns(
-            default_selection=[c.name for c in fields(AutoEvalColumn) if c.displayed_by_default],
-            cant_deselect=[c.name for c in fields(AutoEvalColumn) if c.never_hidden],
-            label="Select Columns to Display:",
-        ),
-        search_columns=[AutoEvalColumn.model.name, AutoEvalColumn.license.name],
-        hide_columns=[c.name for c in fields(AutoEvalColumn) if c.hidden],
-        filter_columns=[
-            ColumnFilter(AutoEvalColumn.model_type.name, type="checkboxgroup", label="Model types"),
-            ColumnFilter(AutoEvalColumn.precision.name, type="checkboxgroup", label="Precision"),
-            ColumnFilter(
-                AutoEvalColumn.params.name,
-                type="slider",
-                min=0.01,
-                max=150,
-                label="Select the number of parameters (B)",
-            ),
-            ColumnFilter(
-                AutoEvalColumn.still_on_hub.name, type="boolean", label="Deleted/incomplete", default=True
-            ),
-        ],
-        bool_checkboxgroup_label="Hide models",
-        interactive=False,
-    )
 demo = gr.Blocks(css=custom_css)
 with demo:
     gr.HTML(TITLE)
     gr.Markdown(INTRODUCTION_TEXT, elem_classes="markdown-text")
     with gr.Tabs(elem_classes="tab-buttons") as tabs:
-        with gr.TabItem("🏅 LLM Benchmark", elem_id="llm-benchmark-tab-table", id=0):
-            leaderboard = init_leaderboard(LEADERBOARD_DF)
-        with gr.TabItem("📝 About", elem_id="llm-benchmark-tab-table", id=2):
-            gr.Markdown(LLM_BENCHMARKS_TEXT, elem_classes="markdown-text")
-        with gr.TabItem("🚀 Submit here! ", elem_id="llm-benchmark-tab-table", id=3):
-            with gr.Column():
-                with gr.Row():
-                    gr.Markdown(EVALUATION_QUEUE_TEXT, elem_classes="markdown-text")
-                with gr.Column():
-                    with gr.Accordion(
-                        f"✅ Finished Evaluations ({len(finished_eval_queue_df)})",
-                        open=False,
-                    ):
-                        with gr.Row():
-                            finished_eval_table = gr.components.Dataframe(
-                                value=finished_eval_queue_df,
-                                headers=EVAL_COLS,
-                                datatype=EVAL_TYPES,
-                                row_count=5,
-                            )
-                    with gr.Accordion(
-                        f"🔄 Running Evaluation Queue ({len(running_eval_queue_df)})",
-                        open=False,
-                    ):
-                        with gr.Row():
-                            running_eval_table = gr.components.Dataframe(
-                                value=running_eval_queue_df,
-                                headers=EVAL_COLS,
-                                datatype=EVAL_TYPES,
-                                row_count=5,
-                            )
-                    with gr.Accordion(
-                        f"⏳ Pending Evaluation Queue ({len(pending_eval_queue_df)})",
-                        open=False,
-                    ):
-                        with gr.Row():
-                            pending_eval_table = gr.components.Dataframe(
-                                value=pending_eval_queue_df,
-                                headers=EVAL_COLS,
-                                datatype=EVAL_TYPES,
-                                row_count=5,
-                            )
-            with gr.Row():
-                gr.Markdown("# ✉️✨ Submit your model here!", elem_classes="markdown-text")
             with gr.Row():
-                with gr.Column():
-                    model_name_textbox = gr.Textbox(label="Model name")
-                    revision_name_textbox = gr.Textbox(label="Revision commit", placeholder="main")
-                    model_type = gr.Dropdown(
-                        choices=[t.to_str(" : ") for t in ModelType if t != ModelType.Unknown],
-                        label="Model type",
-                        multiselect=False,
-                        value=None,
-                        interactive=True,
-                    )
-                with gr.Column():
-                    precision = gr.Dropdown(
-                        choices=[i.value.name for i in Precision if i != Precision.Unknown],
-                        label="Precision",
-                        multiselect=False,
-                        value="float16",
-                        interactive=True,
-                    )
-                    weight_type = gr.Dropdown(
-                        choices=[i.value.name for i in WeightType],
-                        label="Weights type",
-                        multiselect=False,
-                        value="Original",
-                        interactive=True,
-                    )
-                    base_model_name_textbox = gr.Textbox(label="Base model (for delta or adapter weights)")
-            submit_button = gr.Button("Submit Eval")
-            submission_result = gr.Markdown()
-            submit_button.click(
-                add_new_eval,
-                [
-                    model_name_textbox,
-                    base_model_name_textbox,
-                    revision_name_textbox,
-                    precision,
-                    weight_type,
-                    model_type,
-                ],
-                submission_result,
             )
     with gr.Row():
         with gr.Accordion("📙 Citation", open=False):
             citation_button = gr.Textbox(
                 value=CITATION_BUTTON_TEXT,
                 label=CITATION_BUTTON_LABEL,
-                lines=20,
                 elem_id="citation-button",
                 show_copy_button=True,
             )
-scheduler = BackgroundScheduler()
-scheduler.add_job(restart_space, "interval", seconds=1800)
-scheduler.start()
-demo.queue(default_concurrency_limit=40).launch()

 import gradio as gr
 import pandas as pd
 from src.about import (
     CITATION_BUTTON_LABEL,
     CITATION_BUTTON_TEXT,
     INTRODUCTION_TEXT,
     LLM_BENCHMARKS_TEXT,
     TITLE,
 )
 from src.display.css_html_js import custom_css
+# =====================================================
+# Static benchmark data extracted from the paper
+# =====================================================
+def get_qa_leaderboard_df() -> pd.DataFrame:
+    """QA benchmark results from Table 1 of the paper."""
+    data = [
+        {
+            "#": 1,
+            "Model": "Qwen3-max",
+            "Size / Access": "Large / Closed",
+            "Overall Avg": 82.49,
+            "MD-KnowledgeEval": 86.57,
+            "LAMMPS-SyntaxEval": 78.40,
+            "Delta vs 8B": 11.99,
+        },
+        {
+            "#": 2,
+            "Model": "Qwen3-32b",
+            "Size / Access": "32B / Open",
+            "Overall Avg": 77.34,
+            "MD-KnowledgeEval": 81.94,
+            "LAMMPS-SyntaxEval": 72.74,
+            "Delta vs 8B": 6.84,
+        },
+        {
+            "#": 3,
+            "Model": "**MD-Instruct-8B** (Ours)",
+            "Size / Access": "8B / Open",
+            "Overall Avg": 74.67,
+            "MD-KnowledgeEval": 76.89,
+            "LAMMPS-SyntaxEval": 72.45,
+            "Delta vs 8B": 4.17,
+        },
+        {
+            "#": 4,
+            "Model": "Qwen-flash",
+            "Size / Access": "Large / Closed",
+            "Overall Avg": 73.47,
+            "MD-KnowledgeEval": 78.64,
+            "LAMMPS-SyntaxEval": 68.30,
+            "Delta vs 8B": 2.97,
+        },
+        {
+            "#": 5,
+            "Model": "Qwen3-14b",
+            "Size / Access": "14B / Open",
+            "Overall Avg": 72.91,
+            "MD-KnowledgeEval": 77.90,
+            "LAMMPS-SyntaxEval": 67.92,
+            "Delta vs 8B": 2.41,
+        },
+        {
+            "#": 6,
+            "Model": "Qwen3-8b (Baseline)",
+            "Size / Access": "8B / Open",
+            "Overall Avg": 70.50,
+            "MD-KnowledgeEval": 75.15,
+            "LAMMPS-SyntaxEval": 65.84,
+            "Delta vs 8B": 0.00,
+        },
+    ]
+    return pd.DataFrame(data)
+def get_codegen_leaderboard_df() -> pd.DataFrame:
+    """Code generation benchmark results from Figure 2 of the paper.
+    Data extracted from the paper text and figure:
+    - MDAgent2-RUNTIME + MD-Code-8B: ExecSucc@3=37.95%, Code-Score-Human=9.32
+    - Direct Prompting + MD-Code-8B: ExecSucc@3=14.23%, Code-Score-Human=9.29
+    - Other combinations inferred from the figure description.
+    """
+    data = [
+        {
+            "#": 1,
+            "Model": "**MD-Code-8B** (Ours)",
+            "Method": "MDAgent2-RUNTIME",
+            "Exec-Success@3 (%)": 37.95,
+            "Code-Score-Human": 9.32,
+        },
+        {
+            "#": 2,
+            "Model": "Qwen3-32B",
+            "Method": "MDAgent2-RUNTIME",
+            "Exec-Success@3 (%)": 35.87,
+            "Code-Score-Human": 9.18,
+        },
+        {
+            "#": 3,
+            "Model": "Qwen3-8B",
+            "Method": "MDAgent2-RUNTIME",
+            "Exec-Success@3 (%)": 28.63,
+            "Code-Score-Human": 8.75,
+        },
+        {
+            "#": 4,
+            "Model": "**MD-Code-8B** (Ours)",
+            "Method": "MDAgent",
+            "Exec-Success@3 (%)": 27.12,
+            "Code-Score-Human": 9.15,
+        },
+        {
+            "#": 5,
+            "Model": "Qwen3-32B",
+            "Method": "MDAgent",
+            "Exec-Success@3 (%)": 25.44,
+            "Code-Score-Human": 8.96,
+        },
+        {
+            "#": 6,
+            "Model": "Qwen3-8B",
+            "Method": "MDAgent",
+            "Exec-Success@3 (%)": 19.08,
+            "Code-Score-Human": 8.31,
+        },
+        {
+            "#": 7,
+            "Model": "**MD-Code-8B** (Ours)",
+            "Method": "Direct Prompting",
+            "Exec-Success@3 (%)": 14.23,
+            "Code-Score-Human": 9.29,
+        },
+        {
+            "#": 8,
+            "Model": "Qwen3-32B",
+            "Method": "Direct Prompting",
+            "Exec-Success@3 (%)": 12.54,
+            "Code-Score-Human": 8.87,
+        },
+        {
+            "#": 9,
+            "Model": "Qwen3-8B",
+            "Method": "Direct Prompting",
+            "Exec-Success@3 (%)": 7.60,
+            "Code-Score-Human": 7.92,
+        },
+    ]
+    return pd.DataFrame(data)
+# Build DataFrames
+QA_DF = get_qa_leaderboard_df()
+CODEGEN_DF = get_codegen_leaderboard_df()
+# =====================================================
+# Gradio App
+# =====================================================
 demo = gr.Blocks(css=custom_css)
 with demo:
     gr.HTML(TITLE)
     gr.Markdown(INTRODUCTION_TEXT, elem_classes="markdown-text")
     with gr.Tabs(elem_classes="tab-buttons") as tabs:
+        # ---- Tab 1: QA Benchmark ----
+        with gr.TabItem("📝 QA Benchmark", elem_id="qa-benchmark-tab", id=0):
+            gr.Markdown(
+                "### Knowledge & Syntax QA Performance\n"
+                "Performance comparison on MD-KnowledgeEval and LAMMPS-SyntaxEval. "
+                "**Delta vs 8B** indicates improvement over the Qwen3-8B baseline.",
+                elem_classes="markdown-text",
+            )
+            qa_table = gr.Dataframe(
+                value=QA_DF,
+                datatype=["number", "markdown", "str", "number", "number", "number", "number"],
+                interactive=False,
+                elem_id="qa-leaderboard-table",
+            )
+        # ---- Tab 2: Code Generation Benchmark ----
+        with gr.TabItem("💻 Code Generation", elem_id="codegen-benchmark-tab", id=1):
+            gr.Markdown(
+                "### LAMMPS Code Generation Performance\n"
+                "Comparison of Exec-Success@3 and Code-Score-Human across different methods and backbone models. "
+                "**Exec-Success@3**: proportion of tasks with at least one executable candidate out of 3. "
+                "**Code-Score-Human**: expert rating in [0, 10].",
+                elem_classes="markdown-text",
+            )
             with gr.Row():
+                method_filter = gr.Dropdown(
+                    choices=["All", "MDAgent2-RUNTIME", "MDAgent", "Direct Prompting"],
+                    value="All",
+                    label="Filter by Method",
+                    interactive=True,
+                )
+            codegen_table = gr.Dataframe(
+                value=CODEGEN_DF,
+                datatype=["number", "markdown", "str", "number", "number"],
+                interactive=False,
+                elem_id="codegen-leaderboard-table",
             )
+            def filter_codegen(method):
+                if method == "All":
+                    df = CODEGEN_DF.copy()
+                else:
+                    df = CODEGEN_DF[CODEGEN_DF["Method"] == method].copy()
+                df["#"] = range(1, len(df) + 1)
+                return df
+            method_filter.change(filter_codegen, inputs=[method_filter], outputs=[codegen_table])
+        # ---- Tab 3: About ----
+        with gr.TabItem("📖 About", elem_id="about-tab", id=2):
+            gr.Markdown(LLM_BENCHMARKS_TEXT, elem_classes="markdown-text")
     with gr.Row():
         with gr.Accordion("📙 Citation", open=False):
             citation_button = gr.Textbox(
                 value=CITATION_BUTTON_TEXT,
                 label=CITATION_BUTTON_LABEL,
+                lines=10,
                 elem_id="citation-button",
                 show_copy_button=True,
             )
+demo.queue(default_concurrency_limit=40).launch()

requirements.txt CHANGED Viewed

@@ -1,16 +1,2 @@
-APScheduler
-black
-datasets
-gradio
-gradio[oauth]
-gradio_leaderboard==0.0.13
-gradio_client
-huggingface-hub>=0.18.0
-matplotlib
-numpy
 pandas
-python-dateutil
-tqdm
-transformers
-tokenizers>=0.15.0
-sentencepiece


1	+ gradio>=4.0.0









2	pandas

src/about.py CHANGED Viewed

@@ -1,6 +1,7 @@
 from dataclasses import dataclass
 from enum import Enum
 @dataclass
 class Task:
     benchmark: str
@@ -8,65 +9,106 @@ class Task:
     col_name: str
-# Select your tasks here
-# ---------------------------------------------------
-class Tasks(Enum):
-    # task_key in the json file, metric_key in the json file, name to display in the leaderboard
-    task0 = Task("anli_r1", "acc", "ANLI")
-    task1 = Task("logiqa", "acc_norm", "LogiQA")
-NUM_FEWSHOT = 0 # Change with your few shot
-# ---------------------------------------------------
-# Your leaderboard name
-TITLE = """<h1 align="center" id="space-title">Demo leaderboard</h1>"""
-# What does your leaderboard evaluate?
 INTRODUCTION_TEXT = """
-Intro text
 """
-# Which evaluations are you running? how can people reproduce what you have?
-LLM_BENCHMARKS_TEXT = f"""
-## How it works
 ## Reproducibility
-To reproduce our results, here is the commands you can run:
 """
 EVALUATION_QUEUE_TEXT = """
-## Some good practices before submitting a model
-### 1) Make sure you can load your model and tokenizer using AutoClasses:
 ```python
 from transformers import AutoConfig, AutoModel, AutoTokenizer
-config = AutoConfig.from_pretrained("your model name", revision=revision)
-model = AutoModel.from_pretrained("your model name", revision=revision)
-tokenizer = AutoTokenizer.from_pretrained("your model name", revision=revision)
 ```
-If this step fails, follow the error messages to debug your model before submitting it. It's likely your model has been improperly uploaded.
-Note: make sure your model is public!
-Note: if your model needs `use_remote_code=True`, we do not support this option yet but we are working on adding it, stay posted!
-### 2) Convert your model weights to [safetensors](https://huggingface.co/docs/safetensors/index)
-It's a new format for storing weights which is safer and faster to load and use. It will also allow us to add the number of parameters of your model to the `Extended Viewer`!
-### 3) Make sure your model has an open license!
-This is a leaderboard for Open LLMs, and we'd love for as many people as possible to know they can use your model 🤗
-### 4) Fill up your model card
-When we add extra information about models to the leaderboard, it will be automatically taken from the model card
-## In case of model failure
-If your model is displayed in the `FAILED` category, its execution stopped.
-Make sure you have followed the above steps first.
-If everything is done, check you can launch the EleutherAIHarness on your model locally, using the above command without modifications (you can add `--limit` to limit the number of examples per task).
 """
 CITATION_BUTTON_LABEL = "Copy the following snippet to cite these results"
-CITATION_BUTTON_TEXT = r"""
-"""

 from dataclasses import dataclass
 from enum import Enum
 @dataclass
 class Task:
     benchmark: str
     col_name: str
+# QA Benchmark Tasks
+class QATasks(Enum):
+    overall_avg = Task("overall_avg", "score", "Overall Avg")
+    knowledge = Task("knowledge", "score", "MD-KnowledgeEval")
+    syntax = Task("syntax", "score", "LAMMPS-SyntaxEval")
+# Code Generation Benchmark Tasks
+class CodeGenTasks(Enum):
+    exec_success_at_3 = Task("exec_success_at_3", "score", "Exec-Success@3 (%)")
+    code_score_human = Task("code_score_human", "score", "Code-Score-Human")
+NUM_FEWSHOT = 0
+TITLE = """<h1 align="center" id="space-title">MD-EvalBench: Molecular Dynamics LLM Benchmark</h1>"""
 INTRODUCTION_TEXT = """
+**MD-EvalBench** is the first comprehensive benchmark for evaluating Large Language Models in the Molecular Dynamics (MD) domain,
+proposed in the paper *"MDAgent2: Large Language Model for Code Generation and Knowledge Q&A in Molecular Dynamics"*.
+The benchmark consists of three evaluation datasets:
+- **MD-KnowledgeEval** (336 questions): Theoretical knowledge assessment covering interatomic potentials, integration algorithms, equilibrium conditions, and statistical ensembles.
+- **LAMMPS-SyntaxEval** (368 questions): Command and syntax understanding assessment for LAMMPS scripting.
+- **LAMMPS-CodeGenEval** (566 tasks): Automatic code generation quality assessment for executable LAMMPS scripts.
+Models are evaluated on both **Question Answering** (knowledge + syntax) and **Code Generation** (execution success + human scoring) capabilities.
 """
+LLM_BENCHMARKS_TEXT = """
+## Evaluation Protocol
+All experiments are repeated three times and the average results are reported.
+### QA Evaluation (MD-KnowledgeEval + LAMMPS-SyntaxEval)
+- Four question types: single-choice, multiple-choice, fill-in-the-blank, and short-answer
+- Three difficulty levels: Easy, Medium, Hard
+- Score: accuracy percentage (0-100)
+### Code Generation Evaluation (LAMMPS-CodeGenEval)
+- **Exec-Success@3**: Proportion of tasks for which at least one of 3 generated candidates can be successfully executed in LAMMPS
+- **Code-Score-Human**: Subjective rating in [0, 10] by domain experts based on readability, robustness, and physical correctness
+### Evaluation Dimensions for LAMMPS Code
+1. Syntax Correctness
+2. Logical Consistency
+3. Parameter Rationality
+4. Core Logic Accuracy
+5. Logical Completeness
+6. Code Completeness
+7. Result Validity
+8. Physical Soundness
+### Generation Settings
+- **Direct Prompting**: Single prompt without tool integration or execution feedback
+- **MDAgent**: Multi-agent framework with generate-evaluate-rewrite loop (prior work)
+- **MDAgent2-RUNTIME**: Deployable multi-agent system integrating code generation, execution, evaluation, and self-correction
+## Dataset Statistics
+| Dataset | Samples |
+|---------|---------|
+| MD-KnowledgeEval | 336 |
+| LAMMPS-SyntaxEval | 368 |
+| LAMMPS-CodeGenEval | 566 |
 ## Reproducibility
+Models are evaluated using the MD-EvalBench benchmark suite. For detailed methodology, please refer to the paper.
 """
 EVALUATION_QUEUE_TEXT = """
+## Submit your model for evaluation
+### Requirements
+1. Your model must be publicly available on the Hugging Face Hub
+2. Model should be compatible with AutoClasses:
 ```python
 from transformers import AutoConfig, AutoModel, AutoTokenizer
+config = AutoConfig.from_pretrained("your-model-name", revision=revision)
+model = AutoModel.from_pretrained("your-model-name", revision=revision)
+tokenizer = AutoTokenizer.from_pretrained("your-model-name", revision=revision)
 ```
+3. Convert weights to [safetensors](https://huggingface.co/docs/safetensors/index) format
+4. Ensure your model has an open license
+5. Fill up your model card with training details
+### Evaluation Process
+Submitted models will be evaluated on all three MD-EvalBench datasets:
+- MD-KnowledgeEval (knowledge QA)
+- LAMMPS-SyntaxEval (syntax QA)
+- LAMMPS-CodeGenEval (code generation)
 """
 CITATION_BUTTON_LABEL = "Copy the following snippet to cite these results"
+CITATION_BUTTON_TEXT = r"""@article{shi2026mdagent2,
+  title={MDAgent2: Large Language Model for Code Generation and Knowledge Q\&A in Molecular Dynamics},
+  author={Shi, Zhuofan and A, Hubao and Shao, Yufei and Dai, Mengyan and Yu, Yadong and Xiang, Pan and Huang, Dongliang and An, Hongxu and Xin, Chunxiao and Shen, Haiyang and Wang, Zhenyu and Na, Yunshan and Ma, Yun and Huang, Gang and Jing, Xiang},
+  journal={Science China Information Sciences},
+  year={2026}
+}"""

src/display/utils.py CHANGED Viewed

@@ -3,15 +3,13 @@ from enum import Enum
 import pandas as pd
-from src.about import Tasks
 def fields(raw_class):
     return [v for k, v in raw_class.__dict__.items() if k[:2] != "__" and k[-2:] != "__"]
-# These classes are for user facing column names,
-# to avoid having to change them all around the code
-# when a modif is needed
 @dataclass
 class ColumnContent:
     name: str
@@ -20,45 +18,42 @@ class ColumnContent:
     hidden: bool = False
     never_hidden: bool = False
-## Leaderboard columns
-auto_eval_column_dict = []
-# Init
-auto_eval_column_dict.append(["model_type_symbol", ColumnContent, ColumnContent("T", "str", True, never_hidden=True)])
-auto_eval_column_dict.append(["model", ColumnContent, ColumnContent("Model", "markdown", True, never_hidden=True)])
-#Scores
-auto_eval_column_dict.append(["average", ColumnContent, ColumnContent("Average ⬆️", "number", True)])
-for task in Tasks:
-    auto_eval_column_dict.append([task.name, ColumnContent, ColumnContent(task.value.col_name, "number", True)])
-# Model information
-auto_eval_column_dict.append(["model_type", ColumnContent, ColumnContent("Type", "str", False)])
-auto_eval_column_dict.append(["architecture", ColumnContent, ColumnContent("Architecture", "str", False)])
-auto_eval_column_dict.append(["weight_type", ColumnContent, ColumnContent("Weight type", "str", False, True)])
-auto_eval_column_dict.append(["precision", ColumnContent, ColumnContent("Precision", "str", False)])
-auto_eval_column_dict.append(["license", ColumnContent, ColumnContent("Hub License", "str", False)])
-auto_eval_column_dict.append(["params", ColumnContent, ColumnContent("#Params (B)", "number", False)])
-auto_eval_column_dict.append(["likes", ColumnContent, ColumnContent("Hub ❤️", "number", False)])
-auto_eval_column_dict.append(["still_on_hub", ColumnContent, ColumnContent("Available on the hub", "bool", False)])
-auto_eval_column_dict.append(["revision", ColumnContent, ColumnContent("Model sha", "str", False, False)])
-# We use make dataclass to dynamically fill the scores from Tasks
-AutoEvalColumn = make_dataclass("AutoEvalColumn", auto_eval_column_dict, frozen=True)
-## For the queue columns in the submission tab
-@dataclass(frozen=True)
-class EvalQueueColumn:  # Queue column
-    model = ColumnContent("model", "markdown", True)
-    revision = ColumnContent("revision", "str", True)
-    private = ColumnContent("private", "bool", True)
-    precision = ColumnContent("precision", "str", True)
-    weight_type = ColumnContent("weight_type", "str", "Original")
-    status = ColumnContent("status", "str", True)
-## All the model information that we might need
 @dataclass
 class ModelDetails:
     name: str
     display_name: str = ""
-    symbol: str = "" # emoji
 class ModelType(Enum):
@@ -83,11 +78,13 @@ class ModelType(Enum):
             return ModelType.IFT
         return ModelType.Unknown
 class WeightType(Enum):
     Adapter = ModelDetails("Adapter")
     Original = ModelDetails("Original")
     Delta = ModelDetails("Delta")
 class Precision(Enum):
     float16 = ModelDetails("float16")
     bfloat16 = ModelDetails("bfloat16")
@@ -99,12 +96,3 @@ class Precision(Enum):
         if precision in ["torch.bfloat16", "bfloat16"]:
             return Precision.bfloat16
         return Precision.Unknown
-# Column selection
-COLS = [c.name for c in fields(AutoEvalColumn) if not c.hidden]
-EVAL_COLS = [c.name for c in fields(EvalQueueColumn)]
-EVAL_TYPES = [c.type for c in fields(EvalQueueColumn)]
-BENCHMARK_COLS = [t.value.col_name for t in Tasks]

 import pandas as pd
+from src.about import QATasks, CodeGenTasks
 def fields(raw_class):
     return [v for k, v in raw_class.__dict__.items() if k[:2] != "__" and k[-2:] != "__"]
 @dataclass
 class ColumnContent:
     name: str
     hidden: bool = False
     never_hidden: bool = False
+# ---- QA Leaderboard columns ----
+qa_column_dict = []
+qa_column_dict.append(["rank", ColumnContent, ColumnContent("#", "number", True, never_hidden=True)])
+qa_column_dict.append(["model", ColumnContent, ColumnContent("Model", "markdown", True, never_hidden=True)])
+qa_column_dict.append(["size_access", ColumnContent, ColumnContent("Size / Access", "str", True)])
+for task in QATasks:
+    qa_column_dict.append([task.name, ColumnContent, ColumnContent(task.value.col_name, "number", True)])
+qa_column_dict.append(["delta_overall", ColumnContent, ColumnContent("Delta vs 8B", "number", True)])
+QALeaderboardColumn = make_dataclass("QALeaderboardColumn", qa_column_dict, frozen=True)
+QA_COLS = [c.name for c in fields(QALeaderboardColumn) if not c.hidden]
+QA_BENCHMARK_COLS = [t.value.col_name for t in QATasks]
+# ---- Code Generation Leaderboard columns ----
+codegen_column_dict = []
+codegen_column_dict.append(["rank", ColumnContent, ColumnContent("#", "number", True, never_hidden=True)])
+codegen_column_dict.append(["model", ColumnContent, ColumnContent("Model", "markdown", True, never_hidden=True)])
+codegen_column_dict.append(["method", ColumnContent, ColumnContent("Method", "str", True)])
+for task in CodeGenTasks:
+    codegen_column_dict.append([task.name, ColumnContent, ColumnContent(task.value.col_name, "number", True)])
+CodeGenLeaderboardColumn = make_dataclass("CodeGenLeaderboardColumn", codegen_column_dict, frozen=True)
+CODEGEN_COLS = [c.name for c in fields(CodeGenLeaderboardColumn) if not c.hidden]
+CODEGEN_BENCHMARK_COLS = [t.value.col_name for t in CodeGenTasks]
+# ---- Model Types (kept for submission compatibility) ----
 @dataclass
 class ModelDetails:
     name: str
     display_name: str = ""
+    symbol: str = ""
 class ModelType(Enum):
             return ModelType.IFT
         return ModelType.Unknown
 class WeightType(Enum):
     Adapter = ModelDetails("Adapter")
     Original = ModelDetails("Original")
     Delta = ModelDetails("Delta")
 class Precision(Enum):
     float16 = ModelDetails("float16")
     bfloat16 = ModelDetails("bfloat16")
         if precision in ["torch.bfloat16", "bfloat16"]:
             return Precision.bfloat16
         return Precision.Unknown

src/envs.py CHANGED Viewed

@@ -4,22 +4,20 @@ from huggingface_hub import HfApi
 # Info to change for your repository
 # ----------------------------------
-TOKEN = os.environ.get("HF_TOKEN") # A read/write token for your org
-OWNER = "demo-leaderboard-backend" # Change to your org - don't forget to create a results and request dataset, with the correct format!
 # ----------------------------------
-REPO_ID = f"{OWNER}/leaderboard"
-QUEUE_REPO = f"{OWNER}/requests"
-RESULTS_REPO = f"{OWNER}/results"
 # If you setup a cache later, just change HF_HOME
-CACHE_PATH=os.getenv("HF_HOME", ".")
 # Local caches
 EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
 EVAL_RESULTS_PATH = os.path.join(CACHE_PATH, "eval-results")
-EVAL_REQUESTS_PATH_BACKEND = os.path.join(CACHE_PATH, "eval-queue-bk")
-EVAL_RESULTS_PATH_BACKEND = os.path.join(CACHE_PATH, "eval-results-bk")
 API = HfApi(token=TOKEN)

 # Info to change for your repository
 # ----------------------------------
+TOKEN = os.environ.get("HF_TOKEN")  # A read/write token for your org
+OWNER = "MDAgent2"  # Change to your org
 # ----------------------------------
+REPO_ID = f"{OWNER}/MD-EvalBench"
+QUEUE_REPO = f"{OWNER}/md-eval-requests"
+RESULTS_REPO = f"{OWNER}/md-eval-results"
 # If you setup a cache later, just change HF_HOME
+CACHE_PATH = os.getenv("HF_HOME", ".")
 # Local caches
 EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
 EVAL_RESULTS_PATH = os.path.join(CACHE_PATH, "eval-results")
 API = HfApi(token=TOKEN)