| \begin{table*}[t] |
| \centering |
| \small |
| \setlength{\tabcolsep}{5pt} |
| \renewcommand{\arraystretch}{1.08} |
|
|
| \caption{\textbf{Generalization performance on the private hold-out set of ICBCBench.} The private set is not publicly released and is designed to evaluate model generalization and prevent benchmark overfitting. Results are reported on both global (EN) and Chinese (ZH) scenarios across objective and subjective tasks. The best and second-best scores are highlighted in \textbf{bold} and \underline{underline}, respectively. Higher is better for all metrics.} |
| \label{tab:results_private_set} |
|
|
| \begin{tabular}{lcccccc} |
| \toprule |
| \multirow{2}{*}{\textbf{System}} & \multicolumn{3}{c}{\textbf{Global (EN)}} & \multicolumn{3}{c}{\textbf{Chinese (ZH)}} \\ |
| \cmidrule(lr){2-4} \cmidrule(lr){5-7} |
| & \textbf{Objective} & \textbf{Subjective} & \textbf{Overall} & \textbf{Objective} & \textbf{Subjective} & \textbf{Overall} \\ |
| \midrule |
| \rowcolor{gray!20} |
| \multicolumn{7}{c}{\textbf{\textit{Closed}}} \\ |
| Gemini-deep-research & \underline{75.00} & 64.68 & 69.84 & 45.00 & \underline{63.19} & 54.09 \\ |
| OpenAI-o3-deep-research & 55.00 & \textbf{69.05} & 62.02 & 35.00 & 61.86 & 48.43 \\ |
| Kimi-deep-research & 55.00 & 59.57 & 57.28 & 40.00 & 55.24 & 47.62 \\ |
| Doubao-deep-research & 40.00 & 47.17 & 43.59 & 25.00 & 49.47 & 37.23 \\ |
| Perplexity-deep-research & 20.00 & 60.53 & 40.27 & 30.00 & 45.64 & 37.82 \\ |
| GPT-5.5 & 20.00 & 57.00 & 38.50 & 20.00 & 53.93 & 36.97 \\ |
| Claude-opus-4-7 & 5.00 & 62.57 & 33.78 & 20.00 & 58.38 & 39.19 \\ |
| Gemini-3.1-pro-preview & 10.00 & 56.43 & 33.22 & 25.00 & 55.40 & 40.20 \\ |
| Grok-3-deepsearch & 10.00 & 53.03 & 31.52 & 15.00 & 41.89 & 28.45 \\ |
| Qwen-deep-research & 10.00 & 48.30 & 29.15 & 20.00 & 45.18 & 32.59 \\ |
| \midrule |
| \rowcolor{gray!20} |
| \multicolumn{7}{c}{\textbf{\textit{Open}}} \\ |
| OpenClaw(+DeepSeek-V4-Pro) & \textbf{85.00} & 64.83 & \textbf{74.91} & \underline{55.00} & 56.09 & \underline{55.55} \\ |
| DeerFlow(+DeepSeek-V4-Pro) & \underline{75.00} & \underline{68.63} & \underline{71.81} & 40.00 & 56.27 & 48.14 \\ |
| OpenClaw(+GPT-5.5) & 70.00 & 61.63 & 65.81 & \textbf{60.00} & 57.24 & \textbf{58.62} \\ |
| DeerFlow(+GPT-5.5) & 65.00 & 59.80 & 62.40 & 45.00 & 57.67 & 51.34 \\ |
| MiroThinker & 65.00 & 43.30 & 54.15 & 40.00 & 36.18 & 38.09 \\ |
| Jina-deepsearch & 20.00 & 48.60 & 34.30 & 10.00 & 45.37 & 27.68 \\ |
| Kimi-k2.5 & 5.00 & 62.63 & 33.81 & 20.00 & \textbf{64.16} & 42.08 \\ |
| Tongyi-deepresearch-30b-a3b & 0.00 & 46.27 & 23.14 & 5.00 & 36.91 & 20.95 \\ |
| DeepSeek-V4-Pro & 5.00 & 22.50 & 13.75 & 20.00 & 49.47 & 34.73 \\ |
| |
| |
| \bottomrule |
| \end{tabular} |
| \end{table*} |