Hi, I’m ! I’m a PhD student in the University of Washington’s H2Lab, advised by Hannaneh Hajishirzi, and a student researcher at Google DeepMind. My research focuses on post-training for language models: making them more useful to more people, improving them beyond next-token prediction (especially with reinforcement learning), and understanding better data mixtures. I also dabble in alternative approaches to language modelling.
I’m from Sydney and completed my undergraduate studies at the University of Sydney, earning degrees in Arts and IT with majors in Linguistics, Classical Greek, and Computer Science. I also worked with the university’s natural language processing group on multi-hop question answering. During and just after my undergraduate studies, I spent time at the Commonwealth Bank of Australia, a few startups, and Optiver. Before my PhD, I was a predoctoral researcher at AI2 on the AllenNLP team.
If you have questions about my work, academia, software, or research—or just want to chat—feel free to reach out at hamishiv [at] cs [dot] washington [dot] edu. I’m generally happy to answer questions. You can also find me as @hamishivi.
Papers
See below for papers I’ve worked on. You can also check out my Semantic Scholar and Google Scholar profiles.
2026
@misc{ivison2026tmax,
title = {Tmax: A simple recipe for terminal agents},
author = {Ivison*, Hamish and Yin*, Junjie Oscar and Shao, Rulin and Xiao, Teng and Lambert, Nathan and Hajishirzi, Hannaneh},
year = {2026},
eprint = {2606.23321},
archiveprefix = {arXiv},
primaryclass = {cs.CL},
url = {https://arxiv.org/abs/2606.23321},
code = {https://github.com/hamishivi/tmax}
}
@article{xiao2026metareinforcementlearningselfreflectionagentic,
title = {Meta-Reinforcement Learning with Self-Reflection for Agentic Search},
author = {Xiao, Teng and Yuan, Yige and Ivison, Hamish and Zhu, Huaisheng and Brahman, Faeze and Lambert, Nathan and Dasigi, Pradeep and Smith, Noah A. and Hajishirzi, Hannaneh},
year = {2026},
journal = {COLM},
eprint = {2603.11327},
archiveprefix = {arXiv},
primaryclass = {cs.LG},
url = {https://arxiv.org/abs/2603.11327},
code = {https://github.com/tengxiao1/MR-Search}
}
@article{drtulu,
title = {{DR Tulu: Reinforcement Learning with Evolving Rubrics for Deep Research}},
author = {{Rulin Shao*, Akari Asai*, Shannon Shen*, Hamish Ivison*, Varsha Kishore, Jingming Zhuo, Xinran Zhao, Molly Park, Sam Finlayson, David Sontag, Tyler Murray, Sewon Min, Pradeep Dasigi, Luca Soldani, Faeze Brahman, Scott Yih, Sherry Tongshuang Wu, Luke Zettlemoyer, Yoon Kim, Hanna Hajishirzi, Pang Wei Koh}},
year = {2026},
journal = {ICML},
url = {https://www.datocms-assets.com/64837/1763476533-dr_tulu.pdf},
code = {https://github.com/rlresearch/dr-tulu}
}
@article{zeng2025rlve,
title = {{RLVE: Scaling Up Reinforcement Learning for Language Models with Adaptive Verifiable Environments}},
author = {Zeng*, Zhiyuan and Ivison*, Hamish and Wang*, Yiping and Yuan*, Lifan and Li, Shuyue Stella and Ye, Zhuorui and Li, Siting and He, Jacqueline and Zhou, Runlong and Chen, Tong and Zhao, Chenyang and Tsvetkov, Yulia and Du, Simon Shaolei and Jaques, Natasha and Peng, Hao and Koh, Pang Wei and Hajishirzi, Hannaneh},
year = {2026},
journal = {ICML},
url = {https://arxiv.org/abs/2511.07317},
code = {https://github.com/Zhiyuan-Zeng/RLVE}
}
2025
@inproceedings{olmo3-report,
title = {Olmo 3},
author = {{Team OLMo (inc. Hamish Ivison, core contributor)}},
booktitle = {Technical Report, 2025},
year = {2025},
url = {https://allenai.org/blog/olmo3}
}
@article{pyatkin2025generalizing,
title = {{Generalizing Verifiable Instruction Following}},
author = {Pyatkin, Valentina and Malik, Saumya and Graf, Victoria and Ivison, Hamish and Huang, Shengyi and Dasigi, Pradeep and Lambert, Nathan and Hajishirzi, Hannaneh},
year = {2025},
journal = {NeurIPS Datasets and Benchmarks Track},
url = {https://arxiv.org/abs/2507.02833},
code = {https://github.com/allenai/IFBench}
}
@article{geng2025delta,
title = {{The Delta Learning Hypothesis: Preference Tuning on Weak Data can Yield Strong Gains}},
author = {Geng, Scott and Ivison, Hamish and Li, Chun-Liang and Sap, Maarten and Li, Jerry and Krishna, Ranjay and Koh, Pang Wei},
year = {2025},
journal = {COLM},
url = {https://arxiv.org/abs/2507.06187}
}
@article{ivisondata2025,
title = {{Large-Scale Data Selection for Instruction Tuning}},
author = {Ivison, Hamish and Zhang, Muru and Brahman, Faeze and Koh, Pang Wei and Dasigi, Pradeep},
year = {2025},
eprint = {2503.01807},
archiveprefix = {arXiv},
primaryclass = {cs.CL},
url = {https://arxiv.org/abs/2503.01807},
code = {https://github.com/hamishivi/automated-instruction-selection}
}
@article{taeivison2025tess2,
title = {{TESS 2: A Large-Scale Generalist Diffusion Language Model}},
author = {Tae*, Jaesung and Ivison*, Hamish and Kumar, Sachin and Cohan, Arman},
year = {2025},
journal = {ACL},
url = {https://arxiv.org/abs/2502.13917},
code = {https://github.com/hamishivi/tess-2}
}
@article{lambert2024tulu3,
title = {Tülu 3: Pushing Frontiers in Open Language Model Post-Training},
author = {Lambert*, Nathan and Morrison*, Jacob and Pyatkin*, Valentina and Huang*, Shengyi and Ivison*, Hamish and Brahman*, Faeze and Miranda*, Lester James V. and Liu, Alisa and Dziri, Nouha and Lyu, Shane and Gu, Yuling and Malik, Saumya and Graf, Victoria and Hwang, Jena D. and Yang, Jiangjiang and Bras, Ronan Le and Tafjord, Oyvind and Wilhelm, Chris and Soldaini, Luca and Smith, Noah A. and Wang, Yizhong and Dasigi, Pradeep and Hajishirzi, Hannaneh},
year = {2025},
journal = {COLM},
email = {tulu@allenai.org},
url = {https://arxiv.org/abs/2411.15124},
code = {https://github.com/allenai/open-instruct}
}
2024
@article{Poddar2024PersonalizingRL,
title = {Personalizing Reinforcement Learning from Human Feedback with Variational Preference Learning},
author = {Poddar*, Sriyash and Wan*, Yanming and Ivison, Hamish and Gupta, Abhishek and Jaques, Natasha},
year = {2024},
url = {https://arxiv.org/abs/2408.10075},
code = {https://github.com/WEIRDLabUW/vpl_llm},
journal = {NeurIPS}
}
@article{ivison2024unpacking,
title = {Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback},
author = {Ivison, Hamish and Wang, Yizhong and Liu, Jiacheng and Wu, Zeqiu and Pyatkin, Valentina and Lambert, Nathan and Smith, Noah A. and Choi, Yejin and Hajishirzi, Hannaneh},
year = {2024},
eprint = {2406.09279},
journal = {NeurIPS},
url = {https://arxiv.org/abs/2406.09279},
code = {https://github.com/allenai/open-instruct}
}
@article{backtracking,
title = {Backtracking Mathematical Reasoning of Language Models to the Pretraining Data},
author = {Razeghi*, Yasaman and Ivison*, Hamish and Singh, Sameer and Elazar, Yanai},
booktitle = {The Second Tiny Papers Track at ICLR 2024},
year = {2024},
url = {https://openreview.net/pdf?id=otHhLO7GZj}
}
@article{tess,
author = {Mahabadi*, Rabeeh Karimi and Ivison*, Hamish and Tae, Jaesung and Henderson, James and Beltagy, Iz and Peters, Matthew E. and Cohan, Arman},
title = {TESS: Text-to-Text Self-Conditioned Simplex Diffusion},
journal = {EACL},
url = {https://arxiv.org/abs/2305.08379},
year = {2024},
code = {https://github.com/allenai/tess-diffusion}
}
2023
@article{ivison2023camels,
title = {Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2},
author = {Ivison*, Hamish and Wang*, Yizhong and Pyatkin, Valentina and Lambert, Nathan and Peters, Matthew and Dasigi, Pradeep and Jang, Joel and Wadden, David and Smith, Noah A. and Beltagy, Iz and Hajishirzi, Hannaneh},
year = {2023},
url = {https://arxiv.org/abs/2311.10702},
eprint = {2311.10702},
journal = {technical report},
primaryclass = {cs.CL},
code = {https://github.com/allenai/open-instruct}
}
@article{tulu,
title = {How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources},
author = {Wang*, Yizhong and Ivison*, Hamish and Dasigi, Pradeep and Hessel, Jack and Khot, Tushar and Chandu, Khyathi Raghavi and Wadden, David and MacMillan, Kelsey and Smith, Noah A. and Beltagy, Iz and Hajishirzi, Hannaneh},
year = {2023},
url = {https://arxiv.org/abs/2306.04751},
eprint = {2306.04751},
journal = {NeurIPS Datasets and Benchmarks Track},
primaryclass = {cs.CL},
code = {https://github.com/allenai/open-instruct}
}
@article{hint,
author = {Ivison, Hamish and Bhagia, Akshita and Wang, Yizhong and Hajishirzi, Hannaneh and Peters, Matthew},
title = {HINT: Hypernetwork Instruction Tuning for Efficient Zero-Shot Generalisation},
journal = {ACL},
url = {https://arxiv.org/abs/2212.10315},
year = {2023},
code = {https://github.com/allenai/hyper-task-descriptions}
}
@article{deft,
author = {Ivison, Hamish and Smith, Noah A. and Hajishirzi, Hannaneh and Dasigi, Pradeep},
title = {Data-Efficient Finetuning Using Cross-Task Nearest Neighbors},
journal = {Findings of ACL},
code = {https://github.com/allenai/data-efficient-finetuning},
url = {https://arxiv.org/abs/2212.00196},
year = {2023}
}
2022
@article{hyperdecoders,
url = {https://arxiv.org/abs/2203.08304},
author = {Ivison, Hamish and Peters, Matthew E.},
keywords = {Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},
title = {Hyperdecoders: Instance-specific decoders for multi-task NLP},
journal = {Findings of EMNLP},
year = {2022},
code = {https://github.com/allenai/hyperdecoders}
}
2021
@article{localinterp,
author = {Luo*, Siwen and Ivison*, Hamish and Han, Soyeon Caren and Poon, Josiah},
title = {Local Interpretations for Explainable Natural Language Processing:
{A} Survey},
year = {2021},
url = {https://arxiv.org/abs/2103.11072},
journal = {ACM Computing Surveys},
eprint = {2103.11072},
timestamp = {Wed, 24 Mar 2021 15:50:40 +0100}
}
2020
@thesis{thesis,
author = {Ivison, Hamish},
title = {Would you like fries with that? Modular Multi-hop Reasoning},
school = {University of Sydney},
type = {Honours Thesis},
year = {2020},
month = nov,
url = {/assets/static/thesis.pdf}
}