@misc{indiciaeac09e602b93e, title = {Learning diverse attacks on large language models for robust red-teaming and safety tuning}, author = {Seanie Lee and Minsu Kim and Lynn Cherif and David Dobre and Juho Lee and Sung Ju Hwang and Kenji Kawaguchi and Gauthier Gidel and Yoshua Bengio and Esmeralda S. Whitammer and Moksh Jain}, year = {2026}, url = {https://arxiv.org/abs/2405.18540}, note = {Source identifier: 2405.18540} }