@inproceedings{ijcai2026p871, title = {From Human Videos to Robot Manipulation: A Survey on Scalable Vision-Language-Action Learning with Human-Centric Data}, author = {Feng, Zhiyuan and Li, Qixiu and Liang, Huizhi and Yang, Rushuai and Shen, Yichao and Du, Zhiying and Zhang, Zhaowei and Deng, Yu and Zhao, Li and Zhao, Hao and Lu, Zongqing and Mees, Oier and Pollefeys, Marc and Yang, Jiaolong and Guo, Baining}, booktitle = {Proceedings of the Thirty-Fifth International Joint Conference on Artificial Intelligence, {IJCAI-26}}, publisher = {International Joint Conferences on Artificial Intelligence Organization}, editor = {Diego Calvanese}, pages = {7845--7854}, year = {2026}, month = {8}, note = {Survey Track}, doi = {10.24963/ijcai.2026/871}, url = {https://doi.org/10.24963/ijcai.2026/871}, }