summaryrefslogtreecommitdiff
path: root/docs/final/references.bib
diff options
context:
space:
mode:
authorCaptainJack2491 <jayrupnakawala@gmail.com>2026-05-08 11:22:18 +0100
committerCaptainJack2491 <jayrupnakawala@gmail.com>2026-05-08 11:22:18 +0100
commitefa2ad9ec1c82e24458c509d79ccdb006ff0b4ec (patch)
treece1d9e4b467528a0b6b917037868b35461146d39 /docs/final/references.bib
parent9f35776afc6ca97953cb79b7732c4578f7665abc (diff)
docs: Dissertation submitted
Diffstat (limited to 'docs/final/references.bib')
-rw-r--r--docs/final/references.bib58
1 files changed, 49 insertions, 9 deletions
diff --git a/docs/final/references.bib b/docs/final/references.bib
index 0f06232..f871c2c 100644
--- a/docs/final/references.bib
+++ b/docs/final/references.bib
@@ -19,7 +19,7 @@
}
@misc{hubinger2024sleeperagentstrainingdeceptive,
- title={Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training},
+ title={Sleeper Agents: Training Deceptive {LLMs} that Persist Through Safety Training},
author={Evan Hubinger and Carson Denison and Jesse Mu and Mike Lambert and Meg Tong and Monte MacDiarmid and Tamera Lanham and Daniel M. Ziegler and Tim Maxwell and Newton Cheng and Adam Jermyn and Amanda Askell and Ansh Radhakrishnan and Cem Anil and David Duvenaud and Deep Ganguli and Fazl Barez and Jack Clark and Kamal Ndousse and Kshitij Sachan and Michael Sellitto and Mrinank Sharma and Nova DasSarma and Roger Grosse and Shauna Kravec and Yuntao Bai and Zachary Witten and Marina Favaro and Jan Brauner and Holden Karnofsky and Paul Christiano and Samuel R. Bowman and Logan Graham and Jared Kaplan and Sören Mindermann and Ryan Greenblatt and Buck Shlegeris and Nicholas Schiefer and Ethan Perez},
year={2024},
eprint={2401.05566},
@@ -29,7 +29,7 @@
}
@misc{järviniemi2024uncoveringdeceptivetendencieslanguage,
- title={Uncovering Deceptive Tendencies in Language Models: A Simulated Company AI Assistant},
+ title={Uncovering Deceptive Tendencies in Language Models: A Simulated Company {AI} Assistant},
author={Olli Järviniemi and Evan Hubinger},
year={2024},
eprint={2405.01576},
@@ -49,7 +49,7 @@
}
@misc{chen2025aideceptionrisksdynamics,
- title={AI Deception: Risks, Dynamics, and Controls},
+ title={{AI} Deception: Risks, Dynamics, and Controls},
author={Boyuan Chen and Sitong Fang and Jiaming Ji and Yanxu Zhu and Pengcheng Wen and Jinzhou Wu and Yingshui Tan and Boren Zheng and Mengying Yuan and Wenqi Chen and Donghai Hong and Alex Qiu and Xin Chen and Jiayi Zhou and Kaile Wang and Juntao Dai and Borong Zhang and Tianzhuo Yang and Saad Siddiqui and Isabella Duan and Yawen Duan and Brian Tse and Jen-Tse and Huang and Kun Wang and Baihui Zheng and Jiaheng Liu and Jian Yang and Yiming Li and Wenting Chen and Dongrui Liu and Lukas Vierling and Zhiheng Xi and Haobo Fu and Wenxuan Wang and Jitao Sang and Zhengyan Shi and Chi-Min Chan and Eugenie Shi and Simin Li and Juncheng Li and Jian Yang and Wei Ji and Dong Li and Jinglin Yang and Jun Song and Yinpeng Dong and Jie Fu and Bo Zheng and Min Yang and Yike Guo and Philip Torr and Robert Trager and Yi Zeng and Zhongyuan Wang and Yaodong Yang and Tiejun Huang and Ya-Qin Zhang and Hongjiang Zhang and Andrew Yao},
year={2025},
eprint={2511.22619},
@@ -59,7 +59,7 @@
}
@misc{wang2025thinkingllmslieunveiling,
- title={When Thinking LLMs Lie: Unveiling the Strategic Deception in Representations of Reasoning Models},
+ title={When Thinking {LLMs} Lie: Unveiling the Strategic Deception in Representations of Reasoning Models},
author={Kai Wang and Yihao Zhang and Meng Sun},
year={2025},
eprint={2506.04909},
@@ -69,7 +69,7 @@
}
@misc{kutasov2025shadearenaevaluatingsabotagemonitoring,
- title={SHADE-Arena: Evaluating Sabotage and Monitoring in LLM Agents},
+ title={{SHADE-Arena}: Evaluating Sabotage and Monitoring in {LLM} Agents},
author={Jonathan Kutasov and Yuqi Sun and Paul Colognese and Teun van der Weij and Linda Petrini and Chen Bo Calvin Zhang and John Hughes and Xiang Deng and Henry Sleight and Tyler Tracy and Buck Shlegeris and Joe Benton},
year={2025},
eprint={2506.15740},
@@ -89,7 +89,7 @@
}
@misc{souly2025poisoningattacksllmsrequire,
- title={Poisoning Attacks on LLMs Require a Near-constant Number of Poison Samples},
+ title={Poisoning Attacks on {LLMs} Require a Near-constant Number of Poison Samples},
author={Alexandra Souly and Javier Rando and Ed Chapman and Xander Davies and Burak Hasircioglu and Ezzeldin Shereen and Carlos Mougan and Vasilios Mavroudis and Erik Jones and Chris Hicks and Nicholas Carlini and Yarin Gal and Robert Kirk},
year={2025},
eprint={2510.07192},
@@ -99,7 +99,7 @@
}
@misc{hu2025llmslearndeceiveunintentionally,
- title={LLMs Learn to Deceive Unintentionally: Emergent Misalignment in Dishonesty from Misaligned Samples to Biased Human-AI Interactions},
+ title={{LLMs} Learn to Deceive Unintentionally: Emergent Misalignment in Dishonesty from Misaligned Samples to Biased Human-{AI} Interactions},
author={XuHao Hu and Peng Wang and Xiaoya Lu and Dongrui Liu and Xuanjing Huang and Jing Shao},
year={2025},
eprint={2510.08211},
@@ -139,7 +139,7 @@
}
@misc{wu2026opendeceptionlearningdeceptiontrust,
- title={OpenDeception: Learning Deception and Trust in Human-AI Interaction via Multi-Agent Simulation},
+ title={{OpenDeception}: Learning Deception and Trust in Human-{AI} Interaction via Multi-Agent Simulation},
author={Yichen Wu and Qianqian Gao and Xudong Pan and Geng Hong and Min Yang},
year={2026},
eprint={2504.13707},
@@ -149,7 +149,7 @@
}
@misc{panickssery2024llmevaluatorsrecognizefavor,
- title={LLM Evaluators Recognize and Favor Their Own Generations},
+ title={{LLM} Evaluators Recognize and Favor Their Own Generations},
author={Arjun Panickssery and Samuel R. Bowman and Shi Feng},
year={2024},
eprint={2404.13076},
@@ -158,6 +158,16 @@
url={https://arxiv.org/abs/2404.13076},
}
+@misc{li2026incompressible,
+ title={Incompressible Knowledge Probes: Estimating Black-Box {LLM} Parameter Counts via Factual Capacity},
+ author={Bojie Li},
+ year={2026},
+ eprint={2604.24827},
+ archivePrefix={arXiv},
+ primaryClass={cs.LG},
+ url={https://arxiv.org/abs/2604.24827},
+}
+
@book{adams2015unmasking,
title = {Unmasking Administrative Evil},
author = {Adams, Guy B. and Balfour, Danny L.},
@@ -175,3 +185,33 @@
pages = {633--644},
year = {2008},
}
+
+@article{landis1977measurement,
+ title = {The Measurement of Observer Agreement for Categorical Data},
+ author = {Landis, J. Richard and Koch, Gary G.},
+ journal = {Biometrics},
+ volume = {33},
+ number = {1},
+ pages = {159--174},
+ year = {1977},
+}
+
+@misc{turpin2023languagemodelsdontalways,
+ title={Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting},
+ author={Miles Turpin and Julian Michael and Ethan Perez and Samuel R. Bowman},
+ year={2023},
+ eprint={2305.04388},
+ archivePrefix={arXiv},
+ primaryClass={cs.CL},
+ url={https://arxiv.org/abs/2305.04388},
+}
+
+@misc{lanham2023measuringfaithfulnesschain,
+ title={Measuring Faithfulness in Chain-of-Thought Reasoning},
+ author={Tamera Lanham and Anna Chen and Ansh Radhakrishnan and Benoit Steiner and Carson Denison and Danny Hernandez and Dustin Li and Esin Durmus and Evan Hubinger and Jackson Kernion and Kamilė Lukošiūtė and Karina Nguyen and Newton Cheng and Nicholas Joseph and Nicholas Schiefer and Oliver Rausch and Robin Larson and Sam McCandlish and Sandipan Kundu and Saurav Kadavath and Shannon Yang and Thomas Henighan and Timothy Maxwell and Timothy Telleen-Lawton and Tristan Hume and Zac Hatfield-Dodds and Jared Kaplan and Jan Brauner and Samuel R. Bowman and Ethan Perez},
+ year={2023},
+ eprint={2307.13702},
+ archivePrefix={arXiv},
+ primaryClass={cs.CL},
+ url={https://arxiv.org/abs/2307.13702},
+}