Avg@3 (%) of language models on SOOHAK, a mathematician-curated benchmark of research-level math โ over two subsets: Challenge (340 hard items) and Refusal (99 unanswerable items).
@article{son2026soohak,
title={Soohak: A Mathematician-Curated Benchmark for Evaluating Research-level Math Capabilities of LLMs},
author={Son, Guijin and Kim, Seungone and Arnett, Catherine and Ko, Hyunwoo and Lee, Hyein and Kang, Hyeonah and Longxi, Jiang and Yun, Jin and Lee, JungYup and Lee, Kyungmin and others},
journal={arXiv preprint arXiv:2605.09063},
year={2026}
}