Published in: LIPIcs, Volume 394, 20th International Symposium on Empirical Software Engineering and Measurement (ESEM 2026)
Ryan Dang, Noshin Tahsin, and Thomas Zimmermann. Do LLMs Understand Validity? An Empirical Study of Machine‑Generated Threats to Validity in Software Engineering. In 20th International Symposium on Empirical Software Engineering and Measurement (ESEM 2026). Leibniz International Proceedings in Informatics (LIPIcs), Volume 394, pp. 76:1-76:14, Schloss Dagstuhl – Leibniz-Zentrum für Informatik (2026)
@InProceedings{dang_et_al:LIPIcs.ESEM.2026.76,
author = {Dang, Ryan and Tahsin, Noshin and Zimmermann, Thomas},
title = {{Do LLMs Understand Validity? An Empirical Study of Machine‑Generated Threats to Validity in Software Engineering}},
booktitle = {20th International Symposium on Empirical Software Engineering and Measurement (ESEM 2026)},
pages = {76:1--76:14},
series = {Leibniz International Proceedings in Informatics (LIPIcs)},
ISBN = {978-3-95977-450-5},
ISSN = {1868-8969},
year = {2026},
volume = {394},
editor = {Feldt, Robert and Paasivaara, Maria and Mendez, Daniel and Wagner, Stefan and Bar\'{o}n, Marvin Mu\~{n}oz},
publisher = {Schloss Dagstuhl -- Leibniz-Zentrum f{\"u}r Informatik},
address = {Dagstuhl, Germany},
URL = {https://drops.dagstuhl.de/entities/document/10.4230/LIPIcs.ESEM.2026.76},
URN = {urn:nbn:de:0030-drops-280443},
doi = {10.4230/LIPIcs.ESEM.2026.76},
annote = {Keywords: Threats to Validity, Empirical Research, Large Language Models, LLMs, Software Engineering Research, Automated Analysis}
}