@inproceedings{atkinson2026, title = {{Identifying Introspection From the Inside}}, author = {David I. Atkinson and Dillon Plunkett and David Bau}, year = {2026}, booktitle = {COLM 2026}, url = {https://iii.baulab.info} } @inproceedings{binder2024, title = {{Looking Inward: Language Models Can Learn About Themselves by Introspection}}, author = {Felix J. Binder and James Chua and Tomek Korbak and Henry Sleight and John Hughes and Robert Long and Ethan Perez and Miles Turpin and Owain Evans}, year = {2024}, booktitle = {ICLR 2025}, eprint = {2410.13787}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2410.13787} } @misc{sherburn2024, title = {{Can Language Models Explain Their Own Classification Behavior?}}, author = {Dane Sherburn and Bilal Chughtai and Owain Evans}, year = {2024}, howpublished = {arXiv}, eprint = {2405.07436}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2405.07436} } @inproceedings{betley2025, title = {{Tell me about yourself: LLMs are aware of their learned behaviors}}, author = {Jan Betley and Xuchan Bao and Martín Soto and Anna Sztyber-Betley and James Chua and Owain Evans}, year = {2025}, booktitle = {ICLR 2025}, eprint = {2501.11120}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2501.11120} } @misc{comsa2025, title = {{Does It Make Sense to Speak of Introspection in Large Language Models?}}, author = {Iulia M. Comsa and Murray Shanahan}, year = {2025}, howpublished = {arXiv}, eprint = {2506.05068}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2506.05068} } @misc{li2025, title = {{Training Language Models to Explain Their Own Computations}}, author = {Belinda Z. Li and Zifan Carl Guo and Vincent Huang and Jacob Steinhardt and Jacob Andreas}, year = {2025}, howpublished = {arXiv}, eprint = {2511.08579}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.08579} } @misc{lindsey2025, title = {{Emergent Introspective Awareness in Large Language Models}}, author = {Jack Lindsey}, year = {2025}, howpublished = {Transformer Circuits Thread}, eprint = {2601.01828}, archivePrefix = {arXiv}, url = {https://transformer-circuits.pub/2025/introspection/index.html} } @misc{morris2025, title = {{Tests of LLM introspection need to rule out causal bypassing}}, author = {Adam Morris and Dillon Plunkett}, year = {2025}, howpublished = {LessWrong}, url = {https://www.lesswrong.com/posts/LD8yupMtE6btAE3R9/tests-of-llm-introspection-need-to-rule-out-causal-bypassing} } @misc{plunkett2025, title = {{Self-Interpretability: LLMs Can Describe Complex Internal Processes that Drive Their Decisions, and Improve with Training}}, author = {Dillon Plunkett and Adam Morris and Keerthi Reddy and Jorge Morales}, year = {2025}, howpublished = {arXiv}, eprint = {2505.17120}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2505.17120} } @inproceedings{song2025, title = {{Language Models Fail to Introspect About Their Knowledge of Language}}, author = {Siyuan Song and Jennifer Hu and Kyle Mahowald}, year = {2025}, booktitle = {COLM 2025}, eprint = {2503.07513}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2503.07513} } @misc{song2025, title = {{Privileged Self-Access Matters for Introspection in AI}}, author = {Siyuan Song and Harvey Lederman and Jennifer Hu and Kyle Mahowald}, year = {2025}, howpublished = {arXiv}, eprint = {2508.14802}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2508.14802} } @misc{hahami2026, title = {{Detecting the Disturbance: A Nuanced View of Introspective Abilities in LLMs}}, author = {Ely Hahami and Ishaan Sinha and Lavik Jain and Josh Kaplan and Jon Hahami}, year = {2026}, howpublished = {arXiv}, eprint = {2512.12411}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2512.12411} } @misc{pearson, title = {{Latent Introspection: Models Can Detect Prior Concept Injections}}, author = {Theia Pearson-Vogel and Martin Vanek and Raymond Douglas and Jan Kulveit}, year = {2026}, howpublished = {arXiv}, eprint = {2602.20031}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2602.20031} } @misc{berglund2023, title = {{Taken out of context: On measuring situational awareness in LLMs}}, author = {Lukas Berglund and Asa Cooper Stickland and Mikita Balesni and Max Kaufmann and Meg Tong and Tomasz Korbak and Daniel Kokotajlo and Owain Evans}, year = {2023}, howpublished = {arXiv}, eprint = {2309.00667}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2309.00667} } @inproceedings{treutlein2024, title = {{Connecting the Dots: LLMs can Infer and Verbalize Latent Structure from Disparate Training Data}}, author = {Johannes Treutlein and Dami Choi and Jan Betley and Cem Anil and Samuel Marks and Roger Baker Grosse and Owain Evans}, year = {2024}, booktitle = {NeurIPS 2024}, eprint = {2406.14546}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.14546} } @article{bai2025, title = {{Explicitly unbiased large language models still form biased associations}}, author = {Xuechunzi Bai and Angelina Wang and Ilia Sucholutsky and Thomas L. Griffiths}, year = {2025}, journal = {PNAS}, eprint = {2402.04105}, archivePrefix = {arXiv}, doi = {10.1073/pnas.2416228122}, url = {https://pmc.ncbi.nlm.nih.gov/articles/PMC11874501/} } @misc{cywinski2025, title = {{Eliciting Secret Knowledge from Language Models}}, author = {Bartosz Cywiński and Emil Ryd and Rowan Wang and Senthooran Rajamanoharan and Neel Nanda and Arthur Conmy and Samuel Marks}, year = {2025}, howpublished = {arXiv}, eprint = {2510.01070}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2510.01070} } @misc{lindsey2025, title = {{On the Biology of a Large Language Model}}, author = {Jack Lindsey and Wes Gurnee and Emmanuel Ameisen and Brian Chen and Adam Pearce and Nicholas L. Turner and Craig Citro and David Abrahams and Shan Carter and Basil Hosmer and Jonathan Marcus and Michael Sklar and Adly Templeton and Trenton Bricken and Callum McDougall and Hoagy Cunningham and Thomas Henighan and Adam Jermyn and Andy Jones and Andrew Persic and Zhenyi Qi and T. Ben Thompson and Sam Zimmerman and Kelley Rivoire and Thomas Conerly and Chris Olah and Joshua Batson}, year = {2025}, howpublished = {Transformer Circuits Thread}, url = {https://transformer-circuits.pub/2025/attribution-graphs/biology.html} } @misc{wang2025, title = {{Simple Mechanistic Explanations for Out-Of-Context Reasoning}}, author = {Atticus Wang and Joshua Engels and Oliver Clive-Griffin and Senthooran Rajamanoharan and Neel Nanda}, year = {2025}, howpublished = {arXiv}, eprint = {2507.08218}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2507.08218} }