@misc{indiciaeb6f2cf23042b, title = {FailBench: How Reliable are VLMs at Judging Robot Task Success?}, author = {Zaruhi Navasardyan and Tatul Danielyan and Hrant Davtyan}, year = {2026}, url = {https://arxiv.org/abs/2609.03611}, note = {Source identifier: 2609.03611} }