ICBCBench-Leaderboard / results_private_set.tex
Leonnel1220's picture
Upload folder using huggingface_hub
5148820 verified
Raw
History Blame Contribute Delete
4.31 kB
\begin{table*}[t]
\centering
\small
\setlength{\tabcolsep}{5pt}
\renewcommand{\arraystretch}{1.08}
\caption{\textbf{Generalization performance on the private hold-out set of ICBCBench.} The private set is not publicly released and is designed to evaluate model generalization and prevent benchmark overfitting. Results are reported on both global (EN) and Chinese (ZH) scenarios across objective and subjective tasks. The best and second-best scores are highlighted in \textbf{bold} and \underline{underline}, respectively. Higher is better for all metrics.}
\label{tab:results_private_set}
\begin{tabular}{lcccccc}
\toprule
\multirow{2}{*}{\textbf{System}} & \multicolumn{3}{c}{\textbf{Global (EN)}} & \multicolumn{3}{c}{\textbf{Chinese (ZH)}} \\
\cmidrule(lr){2-4} \cmidrule(lr){5-7}
& \textbf{Objective} & \textbf{Subjective} & \textbf{Overall} & \textbf{Objective} & \textbf{Subjective} & \textbf{Overall} \\
\midrule
\rowcolor{gray!20}
\multicolumn{7}{c}{\textbf{\textit{Closed}}} \\
Gemini-deep-research & \underline{75.00} & 64.68 & 69.84 & 45.00 & \underline{63.19} & 54.09 \\
OpenAI-o3-deep-research & 55.00 & \textbf{69.05} & 62.02 & 35.00 & 61.86 & 48.43 \\
Kimi-deep-research & 55.00 & 59.57 & 57.28 & 40.00 & 55.24 & 47.62 \\
Doubao-deep-research & 40.00 & 47.17 & 43.59 & 25.00 & 49.47 & 37.23 \\
Perplexity-deep-research & 20.00 & 60.53 & 40.27 & 30.00 & 45.64 & 37.82 \\
GPT-5.5 & 20.00 & 57.00 & 38.50 & 20.00 & 53.93 & 36.97 \\
Claude-opus-4-7 & 5.00 & 62.57 & 33.78 & 20.00 & 58.38 & 39.19 \\
Gemini-3.1-pro-preview & 10.00 & 56.43 & 33.22 & 25.00 & 55.40 & 40.20 \\
Grok-3-deepsearch & 10.00 & 53.03 & 31.52 & 15.00 & 41.89 & 28.45 \\
Qwen-deep-research & 10.00 & 48.30 & 29.15 & 20.00 & 45.18 & 32.59 \\
\midrule
\rowcolor{gray!20}
\multicolumn{7}{c}{\textbf{\textit{Open}}} \\
OpenClaw(+DeepSeek-V4-Pro) & \textbf{85.00} & 64.83 & \textbf{74.91} & \underline{55.00} & 56.09 & \underline{55.55} \\
DeerFlow(+DeepSeek-V4-Pro) & \underline{75.00} & \underline{68.63} & \underline{71.81} & 40.00 & 56.27 & 48.14 \\
OpenClaw(+GPT-5.5) & 70.00 & 61.63 & 65.81 & \textbf{60.00} & 57.24 & \textbf{58.62} \\
DeerFlow(+GPT-5.5) & 65.00 & 59.80 & 62.40 & 45.00 & 57.67 & 51.34 \\
MiroThinker & 65.00 & 43.30 & 54.15 & 40.00 & 36.18 & 38.09 \\
Jina-deepsearch & 20.00 & 48.60 & 34.30 & 10.00 & 45.37 & 27.68 \\
Kimi-k2.5 & 5.00 & 62.63 & 33.81 & 20.00 & \textbf{64.16} & 42.08 \\
Tongyi-deepresearch-30b-a3b & 0.00 & 46.27 & 23.14 & 5.00 & 36.91 & 20.95 \\
DeepSeek-V4-Pro & 5.00 & 22.50 & 13.75 & 20.00 & 49.47 & 34.73 \\
% OpenClaw(+DeepSeek-V4-Flash) & -- & -- & -- & -- & -- & -- \\
% DeerFlow(+DeepSeek-V4-Flash) & -- & -- & -- & -- & -- & -- \\
\bottomrule
\end{tabular}
\end{table*}