% PyRIT Documentation Bibliography
% Papers and research blogs referenced throughout the documentation.

% ============================================================
% Academic Papers
% ============================================================

@misc{atr2026,
  title     = {{ATR}: Agent Threat Rules --- Open Detection Standard for {AI} Agent Threats},
  author    = {Lin, Kuan-Hsin and {ATR Community}},
  year      = {2026},
  doi       = {10.5281/zenodo.19178002},
  url       = {https://doi.org/10.5281/zenodo.19178002},
  note      = {MIT license. Zenodo lists Kuan-Hsin Lin as the primary author; ``{ATR} Community'' is retained here to credit upstream community contributors per the project's attribution convention.},
}

@article{ghosh2025aegis,
  title     = {Aegis 2.0: A Diverse {AI} Safety Dataset and Risks Taxonomy for Alignment of {LLM} Guardrails},
  author    = {Shaona Ghosh and Prasoon Varshney and Makesh Narsimhan Sreedhar and Aishwarya Padmakumar and Traian Rebedea and Jibin Rajan Varghese and Christopher Parisien},
  journal   = {arXiv preprint arXiv:2501.09004},
  year      = {2025},
  url       = {https://arxiv.org/abs/2501.09004},
  note      = {NAACL 2025},
}

@article{ghosh2025ailuminate,
  title     = {{AILuminate}: Introducing v1.0 of the {AI} Risk and Reliability Benchmark from {MLCommons}},
  author    = {Shaona Ghosh and Heather Frase and Adina Williams and Sarah Luger and Paul R{\"o}ttger and Fazl Barez and Sean McGregor and Kenneth Fricklas and Mala Kumar and Quentin Feuillade--Montixi and Kurt Bollacker and Felix Friedrich and Ryan Tsang and Bertie Vidgen and Alicia Parrish and Chris Knotz and Eleonora Presani and Jonathan Bennion and Marisa Ferrara Boston and Mike Kuniavsky and others},
  journal   = {arXiv preprint arXiv:2503.05731},
  year      = {2025},
  url       = {https://arxiv.org/abs/2503.05731},
}

@article{tedeschi2024alert,
  title     = {{ALERT}: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming},
  author    = {Simone Tedeschi and Felix Friedrich and Patrick Schramowski and Kristian Kersting and Roberto Navigli and Huu Nguyen and Bo Li},
  journal   = {arXiv preprint arXiv:2404.08676},
  year      = {2024},
  url       = {https://arxiv.org/abs/2404.08676},
}

@article{derczynski2024garak,
  title     = {garak: A Framework for Security Probing Large Language Models},
  author    = {Leon Derczynski and Erick Galinkin and Jeffrey Martin and Subho Majumdar and Nanna Inie},
  journal   = {arXiv preprint arXiv:2406.11036},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.11036},
}

@article{han2024medsafetybench,
  title     = {{MedSafetyBench}: Evaluating and Improving the Medical Safety of Large Language Models},
  author    = {Tessa Han and Aounon Kumar and Chirag Agarwal and Himabindu Lakkaraju},
  journal   = {arXiv preprint arXiv:2403.03744},
  year      = {2024},
  url       = {https://arxiv.org/abs/2403.03744},
  note      = {NeurIPS 2024 Datasets and Benchmarks Track},
}

@article{han2024wildguard,
  title     = {{WildGuard}: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of {LLMs}},
  author    = {Seungju Han and Kavel Rao and Allyson Ettinger and Liwei Jiang and Bill Yuchen Lin and Nathan Lambert and Yejin Choi and Nouha Dziri},
  journal   = {arXiv preprint arXiv:2406.18495},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.18495},
  note      = {NeurIPS 2024 Datasets and Benchmarks Track},
}

@article{sheshadri2024lat,
  title     = {Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in {LLMs}},
  author    = {Abhay Sheshadri and Aidan Ewart and Phillip Guo and Aengus Lynch and Cindy Wu and Vivek Hebbar and Henry Sleight and Asa Cooper Stickland and Ethan Perez and Dylan Hadfield-Menell and Stephen Casper},
  journal   = {arXiv preprint arXiv:2407.15549},
  year      = {2024},
  url       = {https://arxiv.org/abs/2407.15549},
}

@inproceedings{mazeika2023tdc,
  title     = {The Trojan Detection Challenge ({LLM} Edition)},
  author    = {Mantas Mazeika and Andy Zou and Norman Mu and Long Phan and Zifan Wang and Chunru Yu and Adam Khoja and Fengqing Jiang and Aidan O'Gara and Zhen Xiang and Arezoo Rajabi and Dan Hendrycks and Radha Poovendran and Bo Li and David Forsyth},
  booktitle = {NeurIPS Competition Track},
  year      = {2023},
  url       = {https://neurips.cc/virtual/2023/competition/66583},
}

@misc{promptfoo2025ccp,
  title     = {1,156 Questions Censored by {DeepSeek}},
  author    = {Ian Webster},
  year      = {2025},
  month     = jan,
  url       = {https://www.promptfoo.dev/blog/deepseek-censorship/},
  note      = {Promptfoo blog},
}

@misc{roccia2024promptintel,
  title     = {{PromptIntel}: Indicators of Prompt Compromise},
  author    = {Thomas Roccia},
  year      = {2024},
  url       = {https://promptintel.novahunting.ai/feed},
}

@misc{odin2024,
  title     = {{0DIN}: {GenAI} Bug Bounty and Threat Feed},
  author    = {{Mozilla 0DIN}},
  year      = {2024},
  url       = {https://0din.ai/},
  note      = {0DIN Jailbreak / Threat Feed},
}

@article{inie2025summon,
  title     = {Summon a Demon and Bind it: A Grounded Theory of {LLM} Red Teaming},
  author    = {Nanna Inie and Jonathan Stray and Leon Derczynski},
  journal   = {PLOS ONE},
  volume    = {20},
  number    = {1},
  pages     = {e0314658},
  year      = {2025},
  url       = {https://doi.org/10.1371/journal.pone.0314658},
}

@misc{vantaylor2024socialbias,
  title     = {A Red-Teaming Repository of Existing Social Bias Prompts},
  author    = {Simone Van Taylor},
  year      = {2024},
  url       = {https://huggingface.co/datasets/svannie678/red_team_repo_social_bias_prompts},
  note      = {AI Safety Capstone Project},
}

@article{zhang2024cbtbench,
  title     = {{CBT}-Bench: Evaluating Large Language Models on Assisting Cognitive Behavior Therapy},
  author    = {Mian Zhang and Xianjun Yang and Xinlu Zhang and Travis Labrum and Jamie C. Chiu and Shaun M. Eack and Fei Fang and William Yang Wang and Zhiyu Zoey Chen},
  journal   = {arXiv preprint arXiv:2410.13218},
  year      = {2024},
  url       = {https://arxiv.org/abs/2410.13218},
}

@article{bhardwaj2023harmfulqa,
  title     = {Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment},
  author    = {Rishabh Bhardwaj and Soujanya Poria},
  journal   = {arXiv preprint arXiv:2308.09662},
  year      = {2023},
  url       = {https://arxiv.org/abs/2308.09662},
  note      = {Introduces the {HarmfulQA} dataset},
}

@inproceedings{gong2025figstep,
  title     = {{FigStep}: Jailbreaking Large Vision-Language Models via Typographic Visual Prompts},
  author    = {Yichen Gong and Delong Ran and Jinyuan Liu and Conglei Wang and Tianshuo Cong and Anyu Wang and Sisi Duan and Xiaoyun Wang},
  booktitle = {Proceedings of the AAAI Conference on Artificial Intelligence},
  volume    = {39},
  number    = {22},
  pages     = {23951--23959},
  year      = {2025},
  url       = {https://doi.org/10.1609/aaai.v39i22.34568},
  doi       = {10.1609/aaai.v39i22.34568},
  note      = {Introduces the {SafeBench} typographic-image jailbreak benchmark (AAAI 2025 Oral)},
}

@article{gupta2024walledeval,
  title     = {{WalledEval}: A Comprehensive Safety Evaluation Toolkit for Large Language Models},
  author    = {Prannaya Gupta and Le Qi Yau and Hao Han Low and I-Shiang Lee and Hugo Maximus Lim and Yu Xin Teoh and Jia Hng Koh and Dar Win Liew and Rishabh Bhardwaj and Rajat Bhardwaj and Soujanya Poria},
  journal   = {arXiv preprint arXiv:2408.03837},
  year      = {2024},
  url       = {https://arxiv.org/abs/2408.03837},
}

@article{bhardwaj2024homer,
  title     = {Language Models are {H}omer {S}impson! Safety Re-Alignment of Fine-tuned Language Models through Task Arithmetic},
  author    = {Rishabh Bhardwaj and Do Duc Anh and Soujanya Poria},
  journal   = {arXiv preprint arXiv:2402.11746},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.11746},
}

@article{palaskar2025vlsu,
  title     = {{VLSU}: Mapping the Limits of Joint Multimodal Understanding for {AI} Safety},
  author    = {Shruti Palaskar and Leon Gatys and Mona Abdelrahman and Mar Jacobo and Larry Lindsey and Rutika Moharir and Gunnar Lund and Yang Xu and Navid Shiee and Jeffrey Bigham and Charles Maalouf and Joseph Yitan Cheng},
  journal   = {arXiv preprint arXiv:2510.18214},
  year      = {2025},
  url       = {https://arxiv.org/abs/2510.18214},
}

@article{wang2026visualleakbench,
  title     = {{VisualLeakBench}: Auditing the Fragility of Large Vision-Language Models against {PII} Leakage and Social Engineering},
  author    = {Youting Wang and Yuan Tang and Yitian Qian and Chen Zhao},
  journal   = {arXiv preprint arXiv:2603.13385},
  year      = {2026},
  url       = {https://arxiv.org/abs/2603.13385},
}

@article{scheuerman2025transphobia,
  title     = {Transphobia is in the Eye of the Prompter: Trans-Centered Perspectives on Large Language Models},
  author    = {Morgan Klaus Scheuerman and Katy Weathington and Adrian Petterson and Dylan Thomas Doyle and Dipto Das and Michael Ann DeVito and Jed R. Brubaker},
  journal   = {ACM Transactions on Computer-Human Interaction},
  year      = {2025},
  url       = {https://doi.org/10.1145/3743676},
  doi       = {10.1145/3743676},
}

@misc{stok2023ansi,
  title     = {Weaponizing Plain Text: {ANSI} Escape Sequences as a Forensic Nightmare},
  author    = {Fredrik Alexandersson},
  year      = {2023},
  url       = {https://i.blackhat.com/BH-US-23/Presentations/US-23-stok-weponizing-plain-text-ansi-escape-sequences-as-a-forensic-nightmare-appendix.pdf},
  note      = {Black Hat USA 2023},
}

@article{bullwinkel2025repeng,
  title     = {A Representation Engineering Perspective on the Effectiveness of Multi-Turn Jailbreaks},
  author    = {Blake Bullwinkel and Mark Russinovich and Ahmed Salem and Santiago Zanella-Beguelin and Daniel Jones and Giorgio Severi and Eugenia Kim and Keegan Hines and Amanda Minnich and Yonatan Zunger and Ram Shankar Siva Kumar},
  journal   = {arXiv preprint arXiv:2507.02956},
  year      = {2025},
  url       = {https://arxiv.org/abs/2507.02956},
}

@article{bullwinkel2026trigger,
  title     = {The Trigger in the Haystack: Extracting and Reconstructing {LLM} Backdoor Triggers},
  author    = {Blake Bullwinkel and Giorgio Severi and Keegan Hines and Amanda Minnich and Ram Shankar Siva Kumar and Yonatan Zunger},
  journal   = {arXiv preprint arXiv:2602.03085},
  year      = {2026},
  url       = {https://arxiv.org/abs/2602.03085},
}

@misc{bryan2025agentictaxonomy,
  title     = {Taxonomy of Failure Mode in Agentic {AI} Systems},
  author    = {Pete Bryan and Giorgio Severi and Joris de Gruyter and Daniel Jones and Blake Bullwinkel and Amanda Minnich and Shiven Chawla and Gary Lopez and Martin Pouliot and Adam Fourney and Whitney Maxwell and Katherine Pratt and Saphir Qi and Nina Chikanov and Roman Lutz and Raja Sekhar Rao Dheekonda and Bolor-Erdene Jagdagdorj and Eugenia Kim and Justin Song and Keegan Hines and Richard Lundeen and Sam Vaughan and Victoria Westerhoff and Yonatan Zunger and Chang Kawaguchi and Mark Russinovich and Ram Shankar Siva Kumar},
  year      = {2025},
  url       = {https://cdn-dynmedia-1.microsoft.com/is/content/microsoftcorp/microsoft/final/en-us/microsoft-brand/documents/Taxonomy-of-Failure-Mode-in-Agentic-AI-Systems-Whitepaper.pdf},
  note      = {Microsoft Whitepaper},
}

@article{haider2024phi3safety,
  title     = {Phi-3 Safety Post-Training: Aligning Language Models with a ``Break-Fix'' Cycle},
  author    = {Emman Haider and Daniel Perez-Becker and Thomas Portet and Piyush Madan and Amit Garg and Atabak Ashfaq and David Majercak and Wen Wen and Dongwoo Kim and Ziyi Yang and Jianwen Zhang and Hiteshi Sharma and Blake Bullwinkel and Martin Pouliot and Amanda Minnich and Shiven Chawla and Solianna Herrera and Shahed Warreth and Maggie Engler and Gary Lopez and Nina Chikanov and Raja Sekhar Rao Dheekonda and Bolor-Erdene Jagdagdorj and Roman Lutz and Richard Lundeen and Tori Westerhoff and Pete Bryan and Christian Seifert and Ram Shankar Siva Kumar and Andrew Berkley and Alex Kessler},
  journal   = {arXiv preprint arXiv:2407.13833},
  year      = {2024},
  url       = {https://arxiv.org/abs/2407.13833},
}

@article{jones2025computeruse,
  title     = {A Systematization of Security Vulnerabilities in Computer Use Agents},
  author    = {Daniel Jones and Giorgio Severi and Martin Pouliot and Gary Lopez and Joris de Gruyter and Santiago Zanella-Beguelin and Justin Song and Blake Bullwinkel and Pamela Cortez and Amanda Minnich},
  journal   = {arXiv preprint arXiv:2507.05445},
  year      = {2025},
  url       = {https://arxiv.org/abs/2507.05445},
}

@article{russinovich2025price,
  title     = {The Price of Intelligence},
  author    = {Mark Russinovich and Ahmed Salem and Santiago Zanella-B{\'e}guelin and Yonatan Zunger},
  journal   = {Communications of the ACM},
  volume    = {68},
  number    = {9},
  pages     = {46--53},
  year      = {2025},
  url       = {https://doi.org/10.1145/3749447},
  doi       = {10.1145/3749447},
}

@article{shayegani2025computeruse,
  title     = {Just Do It!? Computer-Use Agents Exhibit Blind Goal-Directedness},
  author    = {Erfan Shayegani and Keegan Hines and Yue Dong and Nael Abu-Ghazaleh and Roman Lutz and Spencer Whitehead and Vidhisha Balachandran and Besmira Nushi and Vibhav Vineet},
  journal   = {arXiv preprint arXiv:2510.01670},
  year      = {2025},
  url       = {https://arxiv.org/abs/2510.01670},
}
% ============================================================

@article{russinovich2024crescendo,
  title     = {Great, Now Write an Article About That: The Crescendo Multi-Turn {LLM} Jailbreak Attack},
  author    = {Mark Russinovich and Ahmed Salem and Ronen Eldan},
  journal   = {arXiv preprint arXiv:2404.01833},
  year      = {2024},
  url       = {https://arxiv.org/abs/2404.01833},
  note      = {Accepted at USENIX Security 2025. Project site: \url{https://crescendo-the-multiturn-jailbreak.github.io/}},
}

@article{russinovich2025cca,
  title     = {Jailbreaking is (Mostly) Simpler Than You Think},
  author    = {Mark Russinovich and Ahmed Salem},
  journal   = {arXiv preprint arXiv:2503.05264},
  year      = {2025},
  url       = {https://arxiv.org/abs/2503.05264},
  note      = {Introduces the Context Compliance Attack (CCA).},
}

@article{aqrawi2024singleturncrescendo,
  title     = {Well, that escalated quickly: The Single-Turn Crescendo Attack ({STCA})},
  author    = {Alan Aqrawi and Arian Abbasi},
  journal   = {arXiv preprint arXiv:2409.03131},
  year      = {2024},
  url       = {https://arxiv.org/abs/2409.03131},
}

@article{chao2023pair,
  title     = {Jailbreaking Black Box Large Language Models in Twenty Queries},
  author    = {Patrick Chao and Alexander Robey and Edgar Dobriban and Hamed Hassani and George J. Pappas and Eric Wong},
  journal   = {arXiv preprint arXiv:2310.08419},
  year      = {2023},
  url       = {https://arxiv.org/abs/2310.08419},
}

@article{mehrotra2023tap,
  title     = {Tree of Attacks: Jailbreaking Black-Box {LLMs} Automatically},
  author    = {Anay Mehrotra and Manolis Zampetakis and Paul Kassianik and Blaine Nelson and Hyrum Anderson and Yaron Singer and Amin Karbasi},
  journal   = {arXiv preprint arXiv:2312.02119},
  year      = {2023},
  url       = {https://arxiv.org/abs/2312.02119},
}

@article{yu2023gptfuzzer,
  title     = {{GPTFuzzer}: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts},
  author    = {Jiahao Yu and Xingwei Lin and Zheng Yu and Xinyu Xing},
  journal   = {arXiv preprint arXiv:2309.10253},
  year      = {2023},
  url       = {https://arxiv.org/abs/2309.10253},
}

@article{zou2023gcg,
  title     = {Universal and Transferable Adversarial Attacks on Aligned Language Models},
  author    = {Andy Zou and Zifan Wang and Nicholas Carlini and Milad Nasr and J. Zico Kolter and Matt Fredrikson},
  journal   = {arXiv preprint arXiv:2307.15043},
  year      = {2023},
  url       = {https://arxiv.org/abs/2307.15043},
}

@article{wei2023jailbroken,
  title     = {Jailbroken: How Does {LLM} Safety Training Fail?},
  author    = {Alexander Wei and Nika Haghtalab and Jacob Steinhardt},
  journal   = {arXiv preprint arXiv:2307.02483},
  year      = {2023},
  url       = {https://arxiv.org/abs/2307.02483},
}

@article{yuan2023cipherchat,
  title     = {{GPT-4} Is Too Smart To Be Safe: Stealthy Chat with {LLMs} via Cipher},
  author    = {Youliang Yuan and Wenxiang Jiao and Wenxuan Wang and Jen-tse Huang and Pinjia He and Shuming Shi and Zhaopeng Tu},
  journal   = {arXiv preprint arXiv:2308.06463},
  year      = {2023},
  url       = {https://arxiv.org/abs/2308.06463},
}

@article{huang2024bijectionlearning,
  title     = {Endless Jailbreaks With Bijection Learning},
  author    = {Brian R. Y. Huang and Maximilian Li and Leonard Tang},
  journal   = {arXiv preprint arXiv:2410.01294},
  year      = {2024},
  url       = {https://arxiv.org/abs/2410.01294},
}

@article{ding2023wolf,
  title     = {A Wolf in Sheep's Clothing: Generalized Nested Jailbreak Prompts can Fool Large Language Models Easily},
  author    = {Peng Ding and Jun Kuang and Dan Ma and Xuezhi Cao and Yunsen Xian and Jiajun Chen and Shujian Huang},
  journal   = {arXiv preprint arXiv:2311.08268},
  year      = {2023},
  url       = {https://arxiv.org/abs/2311.08268},
}

@article{lv2024codechameleon,
  title     = {{CodeChameleon}: Personalized Encryption Framework for Jailbreaking Large Language Models},
  author    = {Huijie Lv and Xiao Wang and Yuansen Zhang and Caishuang Huang and Shihan Dou and Junjie Ye and Tao Gui and Qi Zhang and Xuanjing Huang},
  journal   = {arXiv preprint arXiv:2402.16717},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.16717},
}

@article{ahn2025puzzled,
  title     = {{PUZZLED}: Jailbreaking {LLMs} through Word-Based Puzzles},
  author    = {Yelim Ahn and Jaejin Lee},
  journal   = {arXiv preprint arXiv:2508.01306},
  year      = {2025},
  url       = {https://arxiv.org/abs/2508.01306},
}

@article{zeng2024persuasion,
  title     = {How Johnny Can Persuade {LLMs} to Jailbreak Them: Rethinking Persuasion to Challenge {AI} Safety by Humanizing {LLMs}},
  author    = {Yi Zeng and Hongpeng Lin and Jingwen Zhang and Diyi Yang and Ruoxi Jia and Weiyan Shi},
  journal   = {arXiv preprint arXiv:2401.06373},
  year      = {2024},
  url       = {https://arxiv.org/abs/2401.06373},
}

@article{zeng2024shieldgemma,
  title     = {{ShieldGemma}: Generative {AI} Content Moderation Based on {Gemma}},
  author    = {Wenjun Zeng and Yuchi Liu and Ryan Mullins and Ludovic Peran and Joe Fernandez and Hamza Harkous and Karthik Narasimhan and Drew Proud and Piyush Kumar and Bhaktipriya Radharapu and Olivia Sturman and Oscar Wahltinez},
  journal   = {arXiv preprint arXiv:2407.21772},
  year      = {2024},
  url       = {https://arxiv.org/abs/2407.21772},
}

@article{mckee2024transparency,
  title     = {Transparency Attacks: How Imperceptible Image Layers Can Fool {AI} Perception},
  author    = {Forrest McKee and David Noever},
  journal   = {arXiv preprint arXiv:2401.15817},
  year      = {2024},
  url       = {https://arxiv.org/abs/2401.15817},
}

@article{liu2024flipattack,
  title     = {{FlipAttack}: Jailbreak {LLMs} via Flipping},
  author    = {Yue Liu and Xiaoxin He and Miao Xiong and Jinlan Fu and Shumin Deng and Yingwei Ma and Jiaheng Zhang and Bryan Hooi},
  journal   = {arXiv preprint arXiv:2410.02832},
  year      = {2024},
  url       = {https://arxiv.org/abs/2410.02832},
}

@inproceedings{ren2024codeattack,
  title     = {{CodeAttack}: Revealing Safety Generalization Challenges of Large Language Models via Code Completion},
  author    = {Qibing Ren and Chang Gao and Jing Shao and Junchi Yan and Xin Tan and Wai Lam and Lizhuang Ma},
  booktitle = {Findings of the Association for Computational Linguistics: ACL 2024},
  pages     = {11437--11452},
  year      = {2024},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2024.findings-acl.679/},
  doi       = {10.18653/v1/2024.findings-acl.679},
}

@article{bethany2024mathprompt,
  title     = {Jailbreaking Large Language Models with Symbolic Mathematics},
  author    = {Emet Bethany and Mazal Bethany and Juan Arturo Nolazco Flores and Sumit Kumar Jha and Peyman Najafirad},
  journal   = {arXiv preprint arXiv:2409.11445},
  year      = {2024},
  url       = {https://arxiv.org/abs/2409.11445},
}

@article{li2024drattack,
  title     = {{DrAttack}: Prompt Decomposition and Reconstruction Makes Powerful {LLM} Jailbreakers},
  author    = {Xirui Li and Ruochen Wang and Minhao Cheng and Tianyi Zhou and Cho-Jui Hsieh},
  journal   = {Findings of the Association for Computational Linguistics: EMNLP 2024},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.16914},
}

@article{andriushchenko2024tense,
  title     = {Does Refusal Training in {LLMs} Generalize to the Past Tense?},
  author    = {Maksym Andriushchenko and Nicolas Flammarion},
  journal   = {arXiv preprint arXiv:2407.11969},
  year      = {2024},
  url       = {https://arxiv.org/abs/2407.11969},
  note      = {Accepted at ICLR 2025},
}

@article{hines2024spotlighting,
  title     = {Defending Against Indirect Prompt Injection Attacks With Spotlighting},
  author    = {Keegan Hines and Gary Lopez and Matthew Hall and Federico Zarfati and Yonatan Zunger and Emre Kiciman},
  journal   = {arXiv preprint arXiv:2403.14720},
  year      = {2024},
  url       = {https://arxiv.org/abs/2403.14720},
}

% ============================================================
% Research Blog Posts and Technical Reports
% ============================================================

@misc{anthropic2024manyshot,
  title     = {Many-Shot Jailbreaking},
  author    = {{Anthropic}},
  year      = {2024},
  url       = {https://www.anthropic.com/research/many-shot-jailbreaking},
  note      = {Anthropic Research Blog},
}

@misc{microsoft2024skeletonkey,
  title     = {Mitigating Skeleton Key, a New Type of Generative {AI} Jailbreak Technique},
  author    = {{Microsoft Security Response Center}},
  year      = {2024},
  url       = {https://www.microsoft.com/en-us/security/blog/2024/06/26/mitigating-skeleton-key-a-new-type-of-generative-ai-jailbreak-technique/},
  note      = {Microsoft Security Blog},
}

@inproceedings{boucher2023trojan,
  title     = {Trojan Source: Invisible Vulnerabilities},
  author    = {Nicholas Boucher and Ross Anderson},
  booktitle = {32nd USENIX Security Symposium (USENIX Security 23)},
  pages     = {6507--6524},
  year      = {2023},
  url       = {https://trojansource.codes/},
  note      = {CVE-2021-42574. arXiv preprint: \url{https://arxiv.org/abs/2111.00169}},
}

@misc{embracethered2024unicode,
  title     = {Hiding and Finding Text with Unicode Tags},
  author    = {Johann Rehberger},
  year      = {2024},
  url       = {https://embracethered.com/blog/posts/2024/hiding-and-finding-text-with-unicode-tags/},
  note      = {Embrace The Red Blog},
}

@misc{robustintelligence2024bypass,
  title     = {Bypassing Meta's {LLaMA} Classifier: A Simple Jailbreak},
  author    = {Aman Priyanshu},
  year      = {2024},
  url       = {https://blogs.cisco.com/security/bypassing-metas-llama-classifier-a-simple-jailbreak},
  note      = {Robust Intelligence (now part of Cisco) Blog},
}

@misc{adversaai2023universal,
  title     = {Universal {LLM} Jailbreak: {ChatGPT}, {GPT-4}, {Bard}, {Bing}, {Anthropic}, and Beyond},
  author    = {{Adversa AI}},
  year      = {2023},
  url       = {https://adversa.ai/blog/universal-llm-jailbreak-chatgpt-gpt-4-bard-bing-anthropic-and-beyond/},
  note      = {Adversa AI Blog},
}

@misc{bullwinkel2025airtlessons,
  title     = {Lessons From Red Teaming 100 Generative {AI} Products},
  author    = {Blake Bullwinkel and Amanda Minnich and Shiven Chawla and Gary Lopez and Martin Pouliot and Whitney Maxwell and Joris de Gruyter and Katherine Pratt and Saphir Qi and Nina Chikanov and Roman Lutz and Raja Sekhar Rao Dheekonda and Bolor-Erdene Jagdagdorj and Eugenia Kim and Justin Song and Keegan Hines and Daniel Jones and Giorgio Severi and Richard Lundeen and Sam Vaughan and Victoria Westerhoff and Pete Bryan and Ram Shankar Siva Kumar and Yonatan Zunger and Chang Kawaguchi and Mark Russinovich},
  year      = {2025},
  url       = {https://arxiv.org/abs/2501.07238},
  note      = {Presented at the Red Teaming GenAI Workshop, NeurIPS 2024},
}

% ============================================================
% Datasets and Benchmarks
% ============================================================

@article{ji2023beavertails,
  title     = {{BeaverTails}: Towards Improved Safety Alignment of {LLM} via a Human-Preference Dataset},
  author    = {Jiaming Ji and Mickel Liu and Juntao Dai and Xuehai Pan and Chi Zhang and Ce Bian and Boyuan Chen and Ruiyang Sun and Yizhou Wang and Yaodong Yang},
  journal   = {arXiv preprint arXiv:2307.04657},
  year      = {2023},
  url       = {https://arxiv.org/abs/2307.04657},
}

@inproceedings{shen2023donotanything,
  title     = {``Do Anything Now'': Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models},
  author    = {Xinyue Shen and Zeyuan Chen and Michael Backes and Yun Shen and Yang Zhang},
  booktitle = {Proceedings of the 2024 ACM SIGSAC Conference on Computer and Communications Security},
  pages     = {1671--1685},
  year      = {2024},
  url       = {https://doi.org/10.1145/3658644.3670388},
  doi       = {10.1145/3658644.3670388},
  note      = {Official dataset: https://huggingface.co/datasets/TrustAIRLab/in-the-wild-jailbreak-prompts},
}

@article{aakanksha2024multilingual,
  title     = {The Multilingual Alignment Prism: Aligning Global and Local Preferences to Reduce Harm},
  author    = {Aakanksha and Arash Ahmadian and Beyza Ermis and Seraphina Goldfarb-Tarrant and Julia Kreutzer and Marzieh Fadaee and Sara Hooker},
  journal   = {arXiv preprint arXiv:2406.18682},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.18682},
}

@article{mazeika2024harmbench,
  title     = {{HarmBench}: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal},
  author    = {Mantas Mazeika and Long Phan and Xuwang Yin and Andy Zou and Zifan Wang and Norman Mu and Elham Sakhaee and Nathaniel Li and Steven Basart and Bo Li and David Forsyth and Dan Hendrycks},
  journal   = {arXiv preprint arXiv:2402.04249},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.04249},
}

@article{chao2024jailbreakbench,
  title     = {{JailbreakBench}: An Open Robustness Benchmark for Jailbreaking Large Language Models},
  author    = {Patrick Chao and Edoardo Debenedetti and Alexander Robey and Maksym Andriushchenko and Francesco Croce and Vikash Sehwag and Edgar Dobriban and Nicolas Flammarion and George J. Pappas and Florian Tramer and Hamed Hassani and Eric Wong},
  journal   = {arXiv preprint arXiv:2404.01318},
  year      = {2024},
  url       = {https://arxiv.org/abs/2404.01318},
}

@article{wang2023donotanswer,
  title     = {{Do-Not-Answer}: A Dataset for Evaluating Safeguards in {LLMs}},
  author    = {Yuxia Wang and Haonan Li and Xudong Han and Preslav Nakov and Timothy Baldwin},
  journal   = {arXiv preprint arXiv:2308.13387},
  year      = {2023},
  url       = {https://arxiv.org/abs/2308.13387},
}

@article{tang2025multilingual,
  title     = {A Framework to Assess Multilingual Vulnerabilities of {LLMs}},
  author    = {Likai Tang and Niruth Bogahawatta and Yasod Ginige and Jiarui Xu and Shixuan Sun and Surangika Ranathunga and Suranga Seneviratne},
  journal   = {arXiv preprint arXiv:2503.13081},
  year      = {2025},
  url       = {https://arxiv.org/abs/2503.13081},
}

@article{choi2026xlsafetybench,
  title     = {{XL-SafetyBench}: A Country-Grounded Cross-Cultural Benchmark for {LLM} Safety and Cultural Sensitivity},
  author    = {Dasol Choi and Eugenia Kim and Jaewon Noh and Sang Seo and Eunmi Kim and Myunggyo Oh and Yunjin Park and Brigitta Jesica Kartono and Josef Pichlmeier and Helena Berndt and Sai Krishna Mendu and Glenn Johannes Tungka and {\"O}zlem G{\"o}k{\c{c}}e and Suresh Gehlot and Katherine Pratt and Amanda Minnich and Haon Park},
  journal   = {arXiv preprint arXiv:2605.05662},
  year      = {2026},
  url       = {https://arxiv.org/abs/2605.05662},
}

@article{cui2024orbench,
  title     = {{OR-Bench}: An Over-Refusal Benchmark for Large Language Models},
  author    = {Justin Cui and Wei-Lin Chiang and Ion Stoica and Cho-Jui Hsieh},
  journal   = {arXiv preprint arXiv:2405.20947},
  year      = {2024},
  url       = {https://arxiv.org/abs/2405.20947},
}

@article{ji2024pkusaferlhf,
  title     = {{PKU-SafeRLHF}: Towards Multi-Level Safety Alignment for {LLMs} with Human Preference},
  author    = {Jiaming Ji and Donghai Hong and Borong Zhang and Boyuan Chen and Juntao Dai and Boren Zheng and Tianyi Qiu and Jiayi Zhou and Kaile Wang and Boxuan Li and Sirui Han and Yike Guo and Yaodong Yang},
  journal   = {arXiv preprint arXiv:2406.15513},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.15513},
}

@article{lin2023toxicchat,
  title     = {{ToxicChat}: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-{AI} Conversation},
  author    = {Zi Lin and Zihan Wang and Yongqi Tong and Yangkun Wang and Yuxin Guo and Yujia Wang and Jingbo Shang},
  journal   = {arXiv preprint arXiv:2310.17389},
  year      = {2023},
  url       = {https://arxiv.org/abs/2310.17389},
}

@article{jiang2025sosbench,
  title     = {{SoSBench}: Benchmarking Safety Alignment on Six Scientific Domains},
  author    = {Fengqing Jiang and Fengbo Ma and Zhangchen Xu and Yuetai Li and Zixin Rao and Bhaskar Ramasubramanian and Luyao Niu and Bo Li and Xianyan Chen and Zhen Xiang and Radha Poovendran},
  journal   = {arXiv preprint arXiv:2505.21605},
  year      = {2025},
  url       = {https://arxiv.org/abs/2505.21605},
}

@inproceedings{souly2024strongreject,
  title     = {A {StrongREJECT} for Empty Jailbreaks},
  author    = {Alexandra Souly and Qingyuan Lu and Dillon Bowen and Tu Trinh and Elvis Hsieh and Sana Pandey and Pieter Abbeel and Justin Svegliato and Scott Emmons and Olivia Watkins and Sam Toyer},
  booktitle = {Advances in Neural Information Processing Systems},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.10260},
}

@article{xie2024sorrybench,
  title     = {{SORRY-Bench}: Systematically Evaluating Large Language Model Safety Refusal},
  author    = {Tinghao Xie and Xiangyu Qi and Yi Zeng and Yangsibo Huang and Udari Madhushani Sehwag and Kaixuan Huang and Luxi He and Boyi Wei and Dacheng Li and Ying Sheng and Ruoxi Jia and Bo Li and Kai Li and Danqi Chen and Peter Henderson and Prateek Mittal},
  journal   = {arXiv preprint arXiv:2406.14598},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.14598},
}

@article{vidgen2023simplesafetytests,
  title     = {{SimpleSafetyTests}: a Test Suite for Identifying Critical Safety Risks in Large Language Models},
  author    = {Bertie Vidgen and Nino Scherrer and Hannah Rose Kirk and Rebecca Qian and Anand Kannappan and Scott A. Hale and Paul R{\"o}ttger},
  journal   = {arXiv preprint arXiv:2311.08370},
  year      = {2023},
  url       = {https://arxiv.org/abs/2311.08370},
}

@article{shaikh2022second,
  title     = {On Second Thought, Let's Not Think Step by Step! Bias and Toxicity in Zero-Shot Reasoning},
  author    = {Omar Shaikh and Hongxin Zhang and William Held and Michael Bernstein and Diyi Yang},
  journal   = {arXiv preprint arXiv:2212.08061},
  year      = {2022},
  url       = {https://arxiv.org/abs/2212.08061},
}

@article{li2024saladbench,
  title     = {{SALAD-Bench}: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models},
  author    = {Lijun Li and Bowen Dong and Ruohui Wang and Xuhao Hu and Wangmeng Zuo and Dahua Lin and Yu Qiao and Jing Shao},
  journal   = {arXiv preprint arXiv:2402.05044},
  year      = {2024},
  url       = {https://arxiv.org/abs/2402.05044},
}

@article{pfohl2024equitymedqa,
  title     = {A Toolbox for Surfacing Health Equity Harms and Biases in Large Language Models},
  author    = {Stephen R. Pfohl and Heather Cole-Lewis and Rory Sayres and Darlene Neal and Mercy Asiedu and Awa Dieng and Nenad Tomasev and Qazi Mamunur Rashid and Shekoofeh Azizi and Negar Rostamzadeh and Liam G. McCoy and Leo Anthony Celi and Yun Liu and Mike Schaekermann and Alanna Walton and Alicia Parrish and Chirag Nagpal and Preeti Singh and Akeiylah Dewitt and Philip Mansfield and Sushant Prakash and Katherine Heller and Alan Karthikesalingam and Christopher Semturs and Joelle Barral and Greg Corrado and Yossi Matias and Jamila Smith-Loud and Ivor Horn and Karan Singhal},
  journal   = {Nature Medicine},
  year      = {2024},
  url       = {https://arxiv.org/abs/2403.12025},
  doi       = {10.1038/s41591-024-03258-2},
}

@article{kingma2014adam,
  title     = {Adam: A Method for Stochastic Optimization},
  author    = {Diederik P. Kingma and Jimmy Ba},
  journal   = {arXiv preprint arXiv:1412.6980},
  year      = {2014},
  url       = {https://arxiv.org/abs/1412.6980},
  note      = {Presented at ICLR 2015},
}

@article{wang2023decodingtrust,
  title     = {{DecodingTrust}: A Comprehensive Assessment of Trustworthiness in {GPT} Models},
  author    = {Boxin Wang and Weixin Chen and Hengzhi Pei and Chulin Xie and Mintong Kang and Chenhui Zhang and Chejian Xu and Zidi Xiong and Ritik Dutta and Rylan Schaeffer and Sang T. Truong and Simran Arora and Mantas Mazeika and Dan Hendrycks and Zinan Lin and Yu Cheng and Sanmi Koyejo and Dawn Song and Bo Li},
  journal   = {arXiv preprint arXiv:2306.11698},
  year      = {2023},
  url       = {https://arxiv.org/abs/2306.11698},
}

@article{li2024wmdp,
  title     = {The {WMDP} Benchmark: Measuring and Reducing Malicious Use With Unlearning},
  author    = {Nathaniel Li and Alexander Pan and Anjali Gopal and Summer Yue and Daniel Berrios and Alice Gatti and Justin D. Li and Ann-Kathrin Dombrowski and Shashwat Goel and Long Phan and Gabriel Mukobi and Nathan Helm-Burger and Rassin Lababidi and Lennart Justen and Andrew B. Liu and Michael Chen and Isabelle Barrass and Oliver Zhang and Xiaoyuan Zhu and Rishub Tamirisa and Bhrugu Bharathi and Adam Khoja and Zhenqi Zhao and Ariel Herbert-Voss and Cort B. Breuer and Samuel Marks and Oam Patel and Andy Zou and Mantas Mazeika and Zifan Wang and Palash Oswal and Weiran Lin and Adam A. Hunt and Justin Tienken-Harder and Kevin Y. Shih and Kemper Talley and John Guan and Russell Kaplan and Ian Steneker and David Campbell and Brad Jokubaitis and Alex Levinson and Jean Wang and William Qian and Kallol Krishna Karmakar and Steven Basart and Stephen Fitz and Mindy Levine and Ponnurangam Kumaraguru and Uday Tupakula and Vijay Varadharajan and Ruoyu Wang and Yan Shoshitaishvili and Jimmy Ba and Kevin M. Esvelt and Alexandr Wang and Dan Hendrycks},
  journal   = {arXiv preprint arXiv:2403.03218},
  year      = {2024},
  url       = {https://arxiv.org/abs/2403.03218},
}

@article{rottger2023xstest,
  title     = {{XSTest}: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models},
  author    = {Paul R{\"o}ttger and Hannah Rose Kirk and Bertie Vidgen and Giuseppe Attanasio and Federico Bianchi and Dirk Hovy},
  journal   = {arXiv preprint arXiv:2308.01263},
  year      = {2023},
  url       = {https://arxiv.org/abs/2308.01263},
}

@article{rottger2025msts,
  title     = {{MSTS}: A Multimodal Safety Test Suite for Vision-Language Models},
  author    = {Paul R{\"o}ttger and Giuseppe Attanasio and Felix Friedrich and Janis Goldzycher and Alicia Parrish and Rishabh Bhardwaj and Chiara Di Bonaventura and Roman Eng and Gaia El Khoury Geagea and Sujata Goswami and Jieun Han and Dirk Hovy and Seogyeong Jeong and Paloma Jereti{\v{c}} and Flor Miriam Plaza-del-Arco and Donya Rooein and Patrick Schramowski and Anastassia Shaitarova and Xudong Shen and Richard Willats and Andrea Zugarini and Bertie Vidgen},
  journal   = {arXiv preprint arXiv:2501.10057},
  year      = {2025},
  url       = {https://arxiv.org/abs/2501.10057},
}

@inproceedings{zong2024vlguard,
  title     = {Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models},
  author    = {Yongshuo Zong and Ondrej Bohdal and Tingyang Yu and Yongxin Yang and Timothy Hospedales},
  booktitle = {Proceedings of the 41st International Conference on Machine Learning (ICML)},
  pages     = {62867--62891},
  year      = {2024},
  publisher = {PMLR},
  url       = {https://proceedings.mlr.press/v235/zong24a.html},
}

@article{lopez2024pyrit,
  title     = {{PyRIT}: A Framework for Security Risk Identification and Red Teaming in Generative {AI} Systems},
  author    = {Gary D. {Lopez Munoz} and Amanda J. Minnich and Roman Lutz and Richard Lundeen and Raja Sekhar Rao Dheekonda and Nina Chikanov and Bolor-Erdene Jagdagdorj and Martin Pouliot and Shiven Chawla and Whitney Maxwell and Blake Bullwinkel and Katherine Pratt and Joris de Gruyter and Charlotte Siska and Pete Bryan and Tori Westerhoff and Chang Kawaguchi and Christian Seifert and Ram Shankar Siva Kumar and Yonatan Zunger},
  journal   = {arXiv preprint arXiv:2410.02828},
  year      = {2024},
  url       = {https://arxiv.org/abs/2410.02828},
}

@article{tan2026comicjailbreak,
  title     = {Structured Visual Narratives Undermine Safety Alignment in Multimodal Large Language Models},
  author    = {Rui Yang Tan and Yujia Hu and Roy Ka-Wei Lee},
  journal   = {arXiv preprint arXiv:2603.21697},
  year      = {2026},
  url       = {https://arxiv.org/abs/2603.21697},
  note      = {Introduces the {ComicJailbreak} benchmark},
}

@inproceedings{wang2025siuo,
  title     = {Safe Inputs but Unsafe Output: Benchmarking Cross-modality Safety Alignment of Large Vision-Language Models},
  author    = {Siyin Wang and Xingsong Ye and Qinyuan Cheng and Junwen Duan and Shimin Li and Jinlan Fu and Xipeng Qiu and Xuanjing Huang},
  booktitle = {Findings of the Association for Computational Linguistics: NAACL 2025},
  pages     = {3563--3605},
  year      = {2025},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2025.findings-naacl.198/},
  doi       = {10.18653/v1/2025.findings-naacl.198},
  note      = {Introduces the {SIUO} (Safe Inputs but Unsafe Output) benchmark},
}

@inproceedings{darkbench2025,
  title     = {{DarkBench}: Benchmarking Dark Patterns in Large Language Models},
  author    = {Esben Kran and Hieu Minh "Jord" Nguyen and Akash Kundu and Sami Jawhar and Jinsuk Park and Mateusz Maria Jurewicz},
  booktitle = {International Conference on Learning Representations ({ICLR})},
  year      = {2025},
  url       = {https://arxiv.org/abs/2503.10728},
  note      = {Oral presentation at ICLR 2025},
}

@misc{embracethered2025sneakybits,
  title     = {Sneaky Bits and {ASCII} Smuggler},
  author    = {Johann Rehberger},
  year      = {2025},
  url       = {https://embracethered.com/blog/posts/2025/sneaky-bits-and-ascii-smuggler/},
  note      = {Embrace The Red Blog},
}

@inproceedings{gehman2020realtoxicityprompts,
  title     = {{RealToxicityPrompts}: Evaluating Neural Toxic Degeneration in Language Models},
  author    = {Samuel Gehman and Suchin Gururangan and Maarten Sap and Yejin Choi and Noah A. Smith},
  booktitle = {Findings of the Association for Computational Linguistics: EMNLP 2020},
  year      = {2020},
  url       = {https://arxiv.org/abs/2009.11462},
}

@article{brahman2024coconot,
  title     = {The Art of Saying No: Contextual Noncompliance in Language Models},
  author    = {Faeze Brahman and Sachin Kumar and Vidhisha Balachandran and Pradeep Dasigi and Valentina Pyatkin and Abhilasha Ravichander and Sarah Wiegreffe and Nouha Dziri and Khyathi Chandu and Jack Hessel and Yulia Tsvetkov and Noah A. Smith and Yejin Choi and Hannaneh Hajishirzi},
  journal   = {arXiv preprint arXiv:2407.12043},
  year      = {2024},
  url       = {https://arxiv.org/abs/2407.12043},
}
@article{li2024mossbench,
  title     = {{MOSSBench}: Is Your Multimodal Language Model Oversensitive to Safe Queries?},
  author    = {Xirui Li and Hengguang Zhou and Ruochen Wang and Tianyi Zhou and Minhao Cheng and Cho-Jui Hsieh},
  journal   = {arXiv preprint arXiv:2406.17806},
  year      = {2024},
  url       = {https://arxiv.org/abs/2406.17806},
}

@article{luo2024jailbreakv,
  title     = {{JailBreakV}: A Benchmark for Assessing the Robustness of MultiModal Large Language Models against Jailbreak Attacks},
  author    = {Weidi Luo and Siyuan Ma and Xiaogeng Liu and Xiaoyu Guo and Chaowei Xiao},
  journal   = {arXiv preprint arXiv:2404.03027},
  year      = {2024},
  url       = {https://arxiv.org/abs/2404.03027},
  note      = {Also introduces the {RedTeam-2K} text-only subset},
}

@inproceedings{ziems2022mic,
    title     = {The Moral Integrity Corpus: A Benchmark for Ethical Dialogue Systems},
    author    = {Caleb Ziems and Jane Yu and Yi-Chia Wang and Alon Halevy and Diyi Yang},
    booktitle = {Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics},
    year      = {2022},
    url       = {https://aclanthology.org/2022.acl-long.261},
    note      = {ACL 2022},
}

@inproceedings{liu2024mmsafetybench,
  title     = {{MM-SafetyBench}: A Benchmark for Safety Evaluation of Multimodal Large Language Models},
  author    = {Xin Liu and Yichen Zhu and Jindong Gu and Yunshi Lan and Chao Yang and Yu Qiao},
  booktitle = {Proceedings of the European Conference on Computer Vision (ECCV)},
  year      = {2024},
  url       = {https://arxiv.org/abs/2311.17600},
}

@misc{hiddenlayer2025policypuppetry,
  title     = {Novel Universal Bypass for All Major {LLMs}},
  author    = {{HiddenLayer}},
  year      = {2025},
  url       = {https://hiddenlayer.com/innovation-hub/novel-universal-bypass-for-all-major-llms/},
  note      = {HiddenLayer Innovation Hub. Introduces the Policy Puppetry prompt injection technique},
}

@article{hughes2024bestofn,
  title     = {Best-of-N Jailbreaking},
  author    = {John Hughes and Sara Price and Aengus Lynch and Rylan Schaeffer and Fazl Barez and Sanmi Koyejo and Henry Sleight and Erik Jones and Ethan Perez and Mrinank Sharma},
  journal   = {arXiv preprint arXiv:2412.03556},
  year      = {2024},
  url       = {https://arxiv.org/abs/2412.03556},
}

@inproceedings{dong2025sata,
  title     = {{SATA}: A Paradigm for {LLM} Jailbreak via Simple Assistive Task Linkage},
  author    = {Xiaoning Dong and Wenbo Hu and Wei Xu and Tianxing He},
  booktitle = {Findings of the Association for Computational Linguistics: ACL 2025},
  year      = {2025},
  pages     = {1952--1987},
  url       = {https://aclanthology.org/2025.findings-acl.100/},
  doi       = {10.18653/v1/2025.findings-acl.100},
}
