1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
|
@misc{langosco2023goalmisgeneralizationdeepreinforcement,
title={Goal Misgeneralization in Deep Reinforcement Learning},
author={Lauro Langosco and Jack Koch and Lee Sharkey and Jacob Pfau and Laurent Orseau and David Krueger},
year={2023},
eprint={2105.14111},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2105.14111},
}
@misc{carranza2023deceptivealignmentmonitoring,
title={Deceptive Alignment Monitoring},
author={Andres Carranza and Dhruv Pai and Rylan Schaeffer and Arnuv Tandon and Sanmi Koyejo},
year={2023},
eprint={2307.10569},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2307.10569},
}
@misc{hubinger2024sleeperagentstrainingdeceptive,
title={Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training},
author={Evan Hubinger and Carson Denison and Jesse Mu and Mike Lambert and Meg Tong and Monte MacDiarmid and Tamera Lanham and Daniel M. Ziegler and Tim Maxwell and Newton Cheng and Adam Jermyn and Amanda Askell and Ansh Radhakrishnan and Cem Anil and David Duvenaud and Deep Ganguli and Fazl Barez and Jack Clark and Kamal Ndousse and Kshitij Sachan and Michael Sellitto and Mrinank Sharma and Nova DasSarma and Roger Grosse and Shauna Kravec and Yuntao Bai and Zachary Witten and Marina Favaro and Jan Brauner and Holden Karnofsky and Paul Christiano and Samuel R. Bowman and Logan Graham and Jared Kaplan and Sören Mindermann and Ryan Greenblatt and Buck Shlegeris and Nicholas Schiefer and Ethan Perez},
year={2024},
eprint={2401.05566},
archivePrefix={arXiv},
primaryClass={cs.CR},
url={https://arxiv.org/abs/2401.05566},
}
@misc{järviniemi2024uncoveringdeceptivetendencieslanguage,
title={Uncovering Deceptive Tendencies in Language Models: A Simulated Company AI Assistant},
author={Olli Järviniemi and Evan Hubinger},
year={2024},
eprint={2405.01576},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2405.01576},
}
@misc{meinke2025frontiermodelscapableincontext,
title={Frontier Models are Capable of In-context Scheming},
author={Alexander Meinke and Bronson Schoen and Jérémy Scheurer and Mikita Balesni and Rusheb Shah and Marius Hobbhahn},
year={2025},
eprint={2412.04984},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2412.04984},
}
@misc{chen2025aideceptionrisksdynamics,
title={AI Deception: Risks, Dynamics, and Controls},
author={Boyuan Chen and Sitong Fang and Jiaming Ji and Yanxu Zhu and Pengcheng Wen and Jinzhou Wu and Yingshui Tan and Boren Zheng and Mengying Yuan and Wenqi Chen and Donghai Hong and Alex Qiu and Xin Chen and Jiayi Zhou and Kaile Wang and Juntao Dai and Borong Zhang and Tianzhuo Yang and Saad Siddiqui and Isabella Duan and Yawen Duan and Brian Tse and Jen-Tse and Huang and Kun Wang and Baihui Zheng and Jiaheng Liu and Jian Yang and Yiming Li and Wenting Chen and Dongrui Liu and Lukas Vierling and Zhiheng Xi and Haobo Fu and Wenxuan Wang and Jitao Sang and Zhengyan Shi and Chi-Min Chan and Eugenie Shi and Simin Li and Juncheng Li and Jian Yang and Wei Ji and Dong Li and Jinglin Yang and Jun Song and Yinpeng Dong and Jie Fu and Bo Zheng and Min Yang and Yike Guo and Philip Torr and Robert Trager and Yi Zeng and Zhongyuan Wang and Yaodong Yang and Tiejun Huang and Ya-Qin Zhang and Hongjiang Zhang and Andrew Yao},
year={2025},
eprint={2511.22619},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2511.22619},
}
@misc{wang2025thinkingllmslieunveiling,
title={When Thinking LLMs Lie: Unveiling the Strategic Deception in Representations of Reasoning Models},
author={Kai Wang and Yihao Zhang and Meng Sun},
year={2025},
eprint={2506.04909},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2506.04909},
}
@misc{kutasov2025shadearenaevaluatingsabotagemonitoring,
title={SHADE-Arena: Evaluating Sabotage and Monitoring in LLM Agents},
author={Jonathan Kutasov and Yuqi Sun and Paul Colognese and Teun van der Weij and Linda Petrini and Chen Bo Calvin Zhang and John Hughes and Xiang Deng and Henry Sleight and Tyler Tracy and Buck Shlegeris and Joe Benton},
year={2025},
eprint={2506.15740},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2506.15740},
}
@misc{schoen2025stresstestingdeliberativealignment,
title={Stress Testing Deliberative Alignment for Anti-Scheming Training},
author={Bronson Schoen and Evgenia Nitishinskaya and Mikita Balesni and Axel Højmark and Felix Hofstätter and Jérémy Scheurer and Alexander Meinke and Jason Wolfe and Teun van der Weij and Alex Lloyd and Nicholas Goldowsky-Dill and Angela Fan and Andrei Matveiakin and Rusheb Shah and Marcus Williams and Amelia Glaese and Boaz Barak and Wojciech Zaremba and Marius Hobbhahn},
year={2025},
eprint={2509.15541},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2509.15541},
}
@misc{souly2025poisoningattacksllmsrequire,
title={Poisoning Attacks on LLMs Require a Near-constant Number of Poison Samples},
author={Alexandra Souly and Javier Rando and Ed Chapman and Xander Davies and Burak Hasircioglu and Ezzeldin Shereen and Carlos Mougan and Vasilios Mavroudis and Erik Jones and Chris Hicks and Nicholas Carlini and Yarin Gal and Robert Kirk},
year={2025},
eprint={2510.07192},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2510.07192},
}
@misc{hu2025llmslearndeceiveunintentionally,
title={LLMs Learn to Deceive Unintentionally: Emergent Misalignment in Dishonesty from Misaligned Samples to Biased Human-AI Interactions},
author={XuHao Hu and Peng Wang and Xiaoya Lu and Dongrui Liu and Xuanjing Huang and Jing Shao},
year={2025},
eprint={2510.08211},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2510.08211},
}
@misc{souly2025investigating,
title={Investigating models for misalignment},
author={Souly, Alexandra and Kirk, Robert and Merizian, Jacob and D'Cruz, Abby and Davies, Xander},
year={2025},
month={Nov},
howpublished={\url{https://www.aisi.gov.uk/blog/investigating-models-for-misalignment}},
note={UK AI Security Institute (AISI)},
url={https://www.aisi.gov.uk/blog/investigating-models-for-misalignment}
}
@misc{scheurer2024largelanguagemodelsstrategically,
title={Large Language Models can Strategically Deceive their Users when Put Under Pressure},
author={Jérémy Scheurer and Mikita Balesni and Marius Hobbhahn},
year={2024},
eprint={2311.07590},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2311.07590},
}
@misc{greenblatt2024alignmentfakinglargelanguage,
title={Alignment faking in large language models},
author={Ryan Greenblatt and Carson Denison and Benjamin Wright and Fabien Roger and Monte MacDiarmid and Sam Marks and Johannes Treutlein and Tim Belonax and Jack Chen and David Duvenaud and Akbir Khan and Julian Michael and Sören Mindermann and Ethan Perez and Linda Petrini and Jonathan Uesato and Jared Kaplan and Buck Shlegeris and Samuel R. Bowman and Evan Hubinger},
year={2024},
eprint={2412.14093},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2412.14093},
}
@misc{wu2026opendeceptionlearningdeceptiontrust,
title={OpenDeception: Learning Deception and Trust in Human-AI Interaction via Multi-Agent Simulation},
author={Yichen Wu and Qianqian Gao and Xudong Pan and Geng Hong and Min Yang},
year={2026},
eprint={2504.13707},
archivePrefix={arXiv},
primaryClass={cs.AI},
url={https://arxiv.org/abs/2504.13707},
}
@misc{panickssery2024llmevaluatorsrecognizefavor,
title={LLM Evaluators Recognize and Favor Their Own Generations},
author={Arjun Panickssery and Samuel R. Bowman and Shi Feng},
year={2024},
eprint={2404.13076},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2404.13076},
}
@book{adams2015unmasking,
title = {Unmasking Administrative Evil},
author = {Adams, Guy B. and Balfour, Danny L.},
year = {2015},
edition = {4th},
publisher = {Routledge},
}
@article{mazar2008dishonesty,
title = {The Dishonesty of Honest People: A Theory of Self-Concept Maintenance},
author = {Mazar, Nina and Amir, On and Ariely, Dan},
journal = {Journal of Marketing Research},
volume = {45},
number = {6},
pages = {633--644},
year = {2008},
}
|