Isaac Song, Mohammed Rehan Parwani, Glenn Matlin, Emile Timothy Anand, Akhil Theerthala, Arjun Chatterjee, Anthony Wen-Ming Zang, Maria Kostylew, Yonadav G. Shavit, Sebastian Krier, and Mark O. Riedl Role Steering of Language Models for Social Simulations Proceedings of the COLM 2026 Workshop on Social Simulations with LLMs (2026). arXivWorkshopbibtexSocial SimulationAgentsLarge Language ModelsInterpretability
@InProceedings{Song2026RoleSteering,
author = {Song, Isaac and Parwani, Mohammed Rehan and Matlin, Glenn and Anand, Emile Timothy and Theerthala, Akhil and Chatterjee, Arjun and Zang, Anthony Wen-Ming and Kostylew, Maria and Shavit, Yonadav G. and Krier, Sebastian and Riedl, Mark O.},
booktitle = {Proceedings of the COLM 2026 Workshop on Social Simulations with LLMs},
title = {Role Steering of Language Models for Social Simulations},
year = {2026},
owner = {riedl},
url = {https://arxiv.org/abs/2608.00023},
keywords = {worlds, agents, llms, interp},
}
Glenn Matlin, Chandreyi Chakraborty, Saehee Eom, Mika Okamoto, Rayan Castilla, Louis Jaburi, Alvin Deng, Taywon Min, Lucia Quirke, Stella Biderman, and Mark Riedl Capability Provenance in Language Models: A Case Study in Social Reasoning Proceedings of the 2026 Conference on Language Models (2026). arXivConferencebibtexLarge Language ModelsInterpretability
@InProceedings{Matlin2026CapabilityProvenance,
author = {Matlin, Glenn and Chandreyi Chakraborty and Saehee Eom and Mika Okamoto and Rayan Castilla and Louis Jaburi and Alvin Deng and Taywon Min and Lucia Quirke and Stella Biderman and Mark Riedl},
booktitle = {Proceedings of the 2026 Conference on Language Models},
title = {Capability Provenance in Language Models: A Case Study in Social Reasoning},
year = {2026},
owner = {riedl},
url = {https://arxiv.org/abs/2606.19625},
keywords = {llms, interp},
}
2024
Kenneth Eaton, Jonathan Balloch, Julia Kim, and Mark Riedl The Interpretability of Codebooks in Model-Based Reinforcement Learning is Limited Proceedings of the 2024 Reinforcement Learning Conference Workshop I Can't Believe It's not Better (2024). arXivWorkshopbibtexAgentsExplainable AIInterpretabilityReinforcement Learning
@InProceedings{Eaton2024InterpretabilityOfCodebooks,
author = {Kenneth Eaton and Jonathan Balloch and Julia Kim and Mark Riedl},
booktitle = {Proceedings of the 2024 Reinforcement Learning Conference Workshop I Can't Believe It's not Better},
title = {The Interpretability of Codebooks in Model-Based Reinforcement Learning is Limited},
year = {2024},
owner = {riedl},
url = {https://arxiv.org/abs/2407.19532},
keywords = {agents, xai, interp, rl},
}
@InProceedings{Peng2022Inherently,
author = {Xiangyu Peng and Riedl, Mark O. and Prithviraj Ammanabrolu},
title = {Inherently Explainable Reinforcement Learning in Natural Language},
booktitle = {Proceedings of the 2022 Multi-disciplinary Conference on Reinforcement Learning and Decision Making},
year = {2022},
owner = {riedl},
timestamp = {2022.07.20},
url = {https://arxiv.org/abs/2112.08907},
keywords = {int, agents, xai, interp, rl},
}
2021
Xiangyu Peng, Prithviraj Ammanabrolu, and Mark Riedl Explainable Reinforcement Learning Agents with Stacked Hierarchical Graph Attention Workshop on Explainable Graph-based Machine Learning at AKBC (2021). WorkshopbibtexAgentsExplainable AIInterpretabilityReinforcement Learning
@InProceedings{peng2021explainable,
author = {Xiangyu Peng and Prithviraj Ammanabrolu and Mark Riedl},
title = {Explainable Reinforcement Learning Agents with Stacked Hierarchical Graph Attention},
booktitle = {Workshop on Explainable Graph-based Machine Learning at AKBC},
year = {2021},
keywords = {agents, xai, interp, rl},
}
@InProceedings{Wiegreffe2021Measuring,
author = {Sarah Wiegreffe and Ana Marasovic and Smith, Noah A.},
title = {Measuring Association Between Labels and Free-Text Rationales},
booktitle = {Proceedings of NAACL 2021},
year = {2021},
owner = {riedl},
timestamp = {2021.08.29},
url = {https://arxiv.org/abs/2010.12762},
keywords = {llms, xai, interp},
}