@inproceedings{ijcai2026p422, title = {StressEval: Failure-Driven Dynamic Benchmarking for Knowledge-Intensive Reasoning in Large Language Models}, author = {Chen, Yongrui and Ma, Yangyang and Huang, Xiaoying and Zhang, Shenyu and Chen, Huajun and Wang, Haofen and Qi, Guilin}, booktitle = {Proceedings of the Thirty-Fifth International Joint Conference on Artificial Intelligence, {IJCAI-26}}, publisher = {International Joint Conferences on Artificial Intelligence Organization}, editor = {Diego Calvanese}, pages = {3792--3800}, year = {2026}, month = {8}, note = {Main Track}, doi = {10.24963/ijcai.2026/422}, url = {https://doi.org/10.24963/ijcai.2026/422}, }