,
Xiangyun Zhan
,
Cheng Wen
,
Bin Yu
,
Ping Chen
,
Xingjian Han
,
Shengchao Qin
Creative Commons Attribution 4.0 International license
Large language models (LLMs) are increasingly used for code translation, yet existing evaluations often collapse results across language pairs, thereby obscuring the direction of transfer. We present an emerging empirical study that treats multilingual code translation as a directed transfer problem. Our study uses 100 programming problems, each with reference implementations in eight languages, yielding a complete directed graph of 56 ordered source-target translation directions. Five contemporary LLMs generate candidate translations, which are evaluated through execution in the target language. The results reveal substantial directional asymmetry. In model-averaged pass@5, reversing a language pair changes the translation success rate by as much as 27.4 percentage points. We also observe systematic differences between source and target roles: Python3 and Java achieve higher average performance when used as source languages, whereas JavaScript, Rust, and Golang achieve higher average performance when used as target languages. These findings show that aggregate scores and unordered language-pair averages can conceal practically important transfer behavior. Rather than establishing a definitive model ranking, this study provides preliminary evidence that code-translation evaluations should preserve ordered source-target results, distinguish source and target language roles, and explicitly report directional gaps. This direction-aware perspective provides a more informative basis for subsequent failure analysis and the evaluation of realistic cross-language migration scenarios.
@InProceedings{wang_et_al:LIPIcs.ESEM.2026.63,
author = {Wang, Quanhe and Zhan, Xiangyun and Wen, Cheng and Yu, Bin and Chen, Ping and Han, Xingjian and Qin, Shengchao},
title = {{A Direction-Aware Study of LLM-Based Code Translation Across Eight Programming Languages}},
booktitle = {20th International Symposium on Empirical Software Engineering and Measurement (ESEM 2026)},
pages = {63:1--63:14},
series = {Leibniz International Proceedings in Informatics (LIPIcs)},
ISBN = {978-3-95977-450-5},
ISSN = {1868-8969},
year = {2026},
volume = {394},
editor = {Feldt, Robert and Paasivaara, Maria and Mendez, Daniel and Wagner, Stefan and Bar\'{o}n, Marvin Mu\~{n}oz},
publisher = {Schloss Dagstuhl -- Leibniz-Zentrum f{\"u}r Informatik},
address = {Dagstuhl, Germany},
URL = {https://drops.dagstuhl.de/entities/document/10.4230/LIPIcs.ESEM.2026.63},
URN = {urn:nbn:de:0030-drops-280313},
doi = {10.4230/LIPIcs.ESEM.2026.63},
annote = {Keywords: Large language models, code translation, multilingual programming, empirical software engineering, execution-based evaluation}
}