I develop foundations for AI Measurement Science, drawing on probabilistic machine learning, measurement theory, and mechanism design to improve how we evaluate AI systems. My work supports the development of AI systems that serve people across diverse backgrounds and needs.
I am advised by Sanmi Koyejo and Nick Haber at the Stanford AI Lab. My research is supported by the Stanford Data Science Scholarship, the Stanford Human-Centered AI Fellowship, and the Microsoft Research Fellowship.
Research
Statistical Foundations of AI Measurement
Building statistical foundations for valid, reliable, and efficient measurement of AI capabilities.
@inproceedings{truong2025reeval,title={Reliable and Efficient Amortized Model-based Evaluation},author={Truong, Sang T. and Tu, Yuheng and Liang, Percy and Li, Bo and Koyejo, Sanmi},booktitle={International Conference on Machine Learning},year={2025},}
Sang T. Truong, Yuheng Tu, Michael Hardy, Anka Reuel, Zeyu Tang, Jirayu Burapacheep, Jonathan Perera, Chibuike Uwakwe, Ben Domingue, Nick Haber, and Sanmi Koyejo
@inproceedings{truong2025fantasticbugs,title={Fantastic Bugs and Where to Find Them in AI Benchmarks},author={Truong, Sang T. and Tu, Yuheng and Hardy, Michael and Reuel, Anka and Tang, Zeyu and Burapacheep, Jirayu and Perera, Jonathan and Uwakwe, Chibuike and Domingue, Ben and Haber, Nick and Koyejo, Sanmi},booktitle={Conference on Neural Information Processing Systems},year={2025},}
Sang Truong, Yuheng Tu, Rylan Schaeffer, and Sanmi Koyejo
@inproceedings{truong2026scalinglaws,title={Item Response Scaling Laws: A Measurement Theory Approach for Efficient and Generalizable Neural Scaling Estimation},author={Truong, Sang and Tu, Yuheng and Schaeffer, Rylan and Koyejo, Sanmi},booktitle={International Conference on Machine Learning},year={2026},}
More publications (7)in Statistical Foundations of AI Measurement
@book{truongkoyejo2026aims,title={AI Measurement Science},author={Truong, Sang T. and Koyejo, Sanmi},year={2026},url={https://aimslab.stanford.edu/textbook/},note={Living textbook},}
Sang T. Truong, Noah Goodman, Emma Brunskill, Ben Domingue, Nick Haber*, and Sanmi Koyejo*
*equal advising
@article{truong2026roadmap,title={A Measurement Science Roadmap: From Human Assessment to AI Evaluation},author={Truong, Sang T. and Goodman, Noah and Brunskill, Emma and Domingue, Ben and Haber, Nick and Koyejo, Sanmi},year={2026},}
Max Zhang, Ameen Patel, Sang T. Truong*, and Sanmi Koyejo*
@inproceedings{zhang2026guardrails,title={Why Do Safety Guardrails Degrade Across Languages?},author={Zhang, Max and Patel, Ameen and Truong, Sang T. and Koyejo, Sanmi},booktitle={Conference on Language Modeling},year={2026},}
COLM 2026
What AI Benchmarks Actually Measure: Adapting Convergent and Discriminant Validity to Interrogate Fifty-Six AI Benchmarks
Meera Desai, Sang T. Truong, Hanna Wallach, Alex Chouldechova, A. Feder Cooper, Jean Garcia-Gathright, Daniel E. Ho, Abigail Z. Jacobs, Sanmi Koyejo, Nicholas Pangakis, and Angelina Wang
@inproceedings{desai2026validity,title={What AI Benchmarks Actually Measure: Adapting Convergent and Discriminant Validity to Interrogate Fifty-Six AI Benchmarks},author={Desai, Meera and Truong, Sang T. and Wallach, Hanna and Chouldechova, Alex and Cooper, A. Feder and Garcia-Gathright, Jean and Ho, Daniel E. and Jacobs, Abigail Z. and Koyejo, Sanmi and Pangakis, Nicholas and Wang, Angelina},booktitle={Conference on Language Modeling},year={2026},}
Michael Hardy, Anka Reuel, Lijin Zhang, Jodi M. Casabianca, Sang Truong, Yash Dave, Hansol Lee, Benjamin Domingue, and Sanmi Koyejo
@inproceedings{hardy2026cartography,title={AI Cartography: Mapping the Latent Landscape of AI Benchmark Ecosystems},author={Hardy, Michael and Reuel, Anka and Zhang, Lijin and Casabianca, Jodi M. and Truong, Sang and Dave, Yash and Lee, Hansol and Domingue, Benjamin and Koyejo, Sanmi},booktitle={International Conference on Machine Learning},year={2026},}
Rodolfo Corona*, Sang Truong*, Ritwik Gupta, Nhi Ngoc Truong, Atnafu Lambebo Tonja, Mena Attia, Fahim Faisal, Kaushal Kumar Maurya, Fred Philippy, Belu Ticona, Sumaya Nur Adan, Fazl Barez, Omar Florez, Supheakmungkol Sarin, Aseem Srivastava, Xiaoyuan Yi, Nick Haber, Dan Klein, Thamar Solorio, Xing Xie, Sanmi Koyejo, and Robert Trager
@inproceedings{corona2026validity,title={Uplifting Human Decision Making in AI Evaluation by Automating Benchmark Validity Analysis},author={Corona, Rodolfo and Truong, Sang and Gupta, Ritwik and Truong, Nhi Ngoc and Tonja, Atnafu Lambebo and Attia, Mena and Faisal, Fahim and Maurya, Kaushal Kumar and Philippy, Fred and Ticona, Belu and Adan, Sumaya Nur and Barez, Fazl and Florez, Omar and Sarin, Supheakmungkol and Srivastava, Aseem and Yi, Xiaoyuan and Haber, Nick and Klein, Dan and Solorio, Thamar and Xie, Xing and Koyejo, Sanmi and Trager, Robert},booktitle={ICML Workshop CTB},year={2026},}
Han Jiang, Susu Zhang, Dongyao Zhu, Yuzhuo Bai, Sang T. Truong, Xiaoyuan Yi, Sanmi Koyejo, Xing Xie, and Ziang Xiao
@article{jiang2026itemlevel,title={AI Evaluation Should Require Standardized Item-Level Data Releases},author={Jiang, Han and Zhang, Susu and Zhu, Dongyao and Bai, Yuzhuo and Truong, Sang T. and Yi, Xiaoyuan and Koyejo, Sanmi and Xie, Xing and Xiao, Ziang},year={2026},}
Incentive-Aware Design of Measurement System
Designing incentives and protocols that encourage informative evaluations and broad capability coverage.
Sang Truong, Serena Wang, Nick Haber, and Sanmi Koyejo
@article{truong2026guardians,title={Strategic Evaluation: Incentivizing AI Capability Coverage with Private Benchmarks},author={Truong, Sang and Wang, Serena and Haber, Nick and Koyejo, Sanmi},year={2026},}
Preprint 2026
Predictive AI Evaluation Competition
Sang Truong, Olawale Salaudeen, Nhi Truong, Serena Wang, Yegor Denisov-Blanch, Fagun Patel, Angelina Wang, Luke Guerdan, Ziang Xiao, Xing Xie, Nick Haber, and Sanmi Koyejo
@article{truong2026predictivecompetition,title={Predictive AI Evaluation Competition},author={Truong, Sang and Salaudeen, Olawale and Truong, Nhi and Wang, Serena and Denisov-Blanch, Yegor and Patel, Fagun and Wang, Angelina and Guerdan, Luke and Xiao, Ziang and Xie, Xing and Haber, Nick and Koyejo, Sanmi},year={2026},}
Preprint 2026
The AI Evaluation Ecosystem
Yash Satish Dave*, Sang T. Truong*, Serena Wang, and Sanmi Koyejo
*Co-first authors
@article{dave2026ecosystem,title={The AI Evaluation Ecosystem},author={Dave, Yash Satish and Truong, Sang T. and Wang, Serena and Koyejo, Sanmi},year={2026},note={*Co-first authors},}
Measuring AI Systems in the Real Worlds
Evaluating AI behavior across languages, communities, and real-world tasks.
Fagun Patel*, Sang T. Truong*, Duc Q. Nguyen*, Kaz Fukuhara, Ben Domingue, Sanmi Koyejo, and Nick Haber
@inproceedings{patel2026iterative,title={A Dataset for Modeling Iterative Problem-Solving},author={Patel, Fagun and Truong, Sang T. and Nguyen, Duc Q. and Fukuhara, Kaz and Domingue, Ben and Koyejo, Sanmi and Haber, Nick},booktitle={Conference on Empirical Methods in Natural Language Processing},year={2026},}
Fagun Patel*, Duc Q. Nguyen*, Sang T. Truong*, Jody Vaynshtok, Sanmi Koyejo, and Nick Haber
@inproceedings{patel2025syntax,title={The Sound of Syntax: Fine-tuning and Comprehensive Evaluation of Language Models for Speech Pathology},author={Patel, Fagun and Nguyen, Duc Q. and Truong, Sang T. and Vaynshtok, Jody and Koyejo, Sanmi and Haber, Nick},booktitle={Conference on Empirical Methods in Natural Language Processing},year={2025},}
@inproceedings{truong2024villm,title={Crossing Linguistic Horizons: Fine-tuning and Comprehensive Evaluation of Vietnamese Large Language Models},author={Truong, Sang T. and Nguyen, Duc Q. and Nguyen, Toan and Le, Dong D. and Truong, Nhi N. and Quan, Tho and Koyejo, Sanmi},booktitle={Findings of the Association for Computational Linguistics: NAACL},year={2024},}
More publications (3)in Measuring AI Systems in the Real Worlds
Zeyu Tang*, Sang T. Truong*, Deonna Owens*, Shreyas Sharma, Yibo Jacky Zhang, Brando Miranda, and Sanmi Koyejo
@inproceedings{tang2026insitu,title={In-Situ Behavioral Evaluation for LLM Fairness, Not Standardized-Test Scores},author={Tang, Zeyu and Truong, Sang T. and Owens, Deonna and Sharma, Shreyas and Zhang, Yibo Jacky and Miranda, Brando and Koyejo, Sanmi},booktitle={Conference on Language Modeling},year={2026},}
@inproceedings{hua2025researchcodebench,title={ResearchCodeBench: Benchmarking LLMs on Implementing Novel Machine Learning Research Code},author={Hua, Tianyu and Hua, Harper and Xiang, Violet and Klieger, Benjamin and Truong, Sang T. and Liang, Weixin and Sun, Fan-Yun and Haber, Nick},booktitle={Conference on Neural Information Processing Systems},year={2025},}
@inproceedings{wang2023decodingtrust,title={DecodingTrust: A Comprehensive Assessment of Trustworthiness in GPT Models},author={Wang, Boxin and Chen, Weixin and Pei, Hengzhi and Xie, Chulin and Kang, Mintong and Zhang, Chenhui and Xu, Chejian and Xiong, Zidi and Dutta, Ritik and Schaeffer, Rylan and Truong, Sang T. and Arora, Simran and Mazeika, Mantas and Hendrycks, Dan and Lin, Zinan and Cheng, Yu and Koyejo, Sanmi and Song, Dawn and Li, Bo},booktitle={Advances in Neural Information Processing Systems},volume={36},year={2023},}