{"url":"/task/code-translation","name":"Code Translation","slug":"code-translation","description_markdown":"Code translation is the process of converting code written in one programming language to another programming language while maintaining the same functionality. This process is also known as code conversion, source-to-source translation, or transpilation. Code translation is often performed when developers want to take advantage of new programming languages, improve code performance, or maintain legacy systems. Some common examples include translating code from Python to Java, or from JavaScript to TypeScript.","categories":[{"name":"Computer Code","url":"/area/computer-code"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":100,"papers_with_code":54,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/code-translation-on-codexglue-codetrans","slug":"code-translation-on-codexglue-codetrans","dataset":"CodeXGLUE - CodeTrans","dataset_url":"/dataset/codexglue","rows_in_archive":2,"metrics":["Accuracy (C#→Java)","Accuracy (Java→C#)","BLEU (C#→Java)","BLEU (Java→C#)","CodeBLEU (C#→Java)","CodeBLEU (Java→C#)"],"first_row_in_archive_order":{"model":"CodeT5","paper_title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","paper_url":"/paper/codet5-identifier-aware-unified-pre-trained","paper_date":"2021-09-02","arxiv_id":"2109.00859","code_links":[{"title":"salesforce/codet5","url":"https://github.com/salesforce/codet5"},{"title":"salesforce/coderl","url":"https://github.com/salesforce/coderl"},{"title":"awsm-research/vulrepair","url":"https://github.com/awsm-research/vulrepair"},{"title":"jetbrains-research/commit_message_generation","url":"https://github.com/jetbrains-research/commit_message_generation"},{"title":"fewshotcdcs/cdcs","url":"https://github.com/fewshotcdcs/cdcs"}],"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}}},{"leaderboard":"/sota/code-translation-on-nlc2cmd","slug":"code-translation-on-nlc2cmd","dataset":"NLC2CMD","dataset_url":"/dataset/nlc2cmd","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ChatGPT","paper_title":"NL2CMD: An Updated Workflow for Natural Language to Bash Commands Translation","paper_url":"/paper/nl2cmd-an-updated-workflow-for-natural","paper_date":"2023-02-15","arxiv_id":"2302.07845","code_links":[{"title":"magnumresearchgroup/magnum-nlc2cmd","url":"https://github.com/magnumresearchgroup/magnum-nlc2cmd"},{"title":"magnumresearchgroup/bash_gen","url":"https://github.com/magnumresearchgroup/bash_gen"}],"syntology":null}}],"datasets":[{"url":"/dataset/codexglue","name":"CodeXGLUE","full_name":"","num_papers_in_archive":205},{"url":"/dataset/humaneval-x","name":"HumanEval-X","full_name":"","num_papers_in_archive":16},{"url":"/dataset/xcodeeval","name":"xCodeEval","full_name":"xCodeEval","num_papers_in_archive":15},{"url":"/dataset/codetransocean","name":"CodeTransOcean","full_name":"","num_papers_in_archive":9},{"url":"/dataset/nlc2cmd","name":"NLC2CMD","full_name":"","num_papers_in_archive":8},{"url":"/dataset/pytorrent","name":"PyTorrent","full_name":"PyTorrent","num_papers_in_archive":5},{"url":"/dataset/sltrans","name":"SLTrans","full_name":"","num_papers_in_archive":2},{"url":"/dataset/crust-bench","name":"CRUST-bench","full_name":"","num_papers_in_archive":1},{"url":"/dataset/fixeval","name":"FixEval","full_name":"FixEval","num_papers_in_archive":1},{"url":"/dataset/obscurax","name":"ObscuraX","full_name":"ObscuraX","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/code-generation","name":"Code Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":54,"tagged_in_all":100,"items":[{"url":"/paper/unsupervised-translation-of-programming","title":"Unsupervised Translation of Programming Languages","date":"2020-06-05","arxiv_id":"2006.03511","repositories_listed":9,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/codexglue-a-machine-learning-benchmark","title":"CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation","date":"2021-02-09","arxiv_id":"2102.04664","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/codet5-identifier-aware-unified-pre-trained","title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","date":"2021-09-02","arxiv_id":"2109.00859","repositories_listed":5,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/codebleu-a-method-for-automatic-evaluation-of","title":"CodeBLEU: a Method for Automatic Evaluation of Code Synthesis","date":"2020-09-22","arxiv_id":"2009.10297","repositories_listed":3,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/mpi-rical-data-driven-mpi-distributed","title":"MPI-rical: Data-Driven MPI Distributed Parallelism Assistance with Transformers","date":"2023-05-16","arxiv_id":"2305.09438","repositories_listed":2,"syntology":null},{"url":"/paper/nl2cmd-an-updated-workflow-for-natural","title":"NL2CMD: An Updated Workflow for Natural Language to Bash Commands Translation","date":"2023-02-15","arxiv_id":"2302.07845","repositories_listed":2,"syntology":null},{"url":"/paper/multi-lingual-evaluation-of-code-generation","title":"Multi-lingual Evaluation of Code Generation Models","date":"2022-10-26","arxiv_id":"2210.14868","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/natgen-generative-pre-training-by","title":"NatGen: Generative pre-training by \"Naturalizing\" source code","date":"2022-06-15","arxiv_id":"2206.07585","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/codeattack-code-based-adversarial-attacks-for","title":"CodeAttack: Code-Based Adversarial Attacks for Pre-trained Programming Language Models","date":"2022-05-31","arxiv_id":"2206.00052","repositories_listed":2,"syntology":null},{"url":"/paper/the-impact-of-lexical-and-grammatical-1","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2022-02-28","arxiv_id":"2202.13972","repositories_listed":2,"syntology":null},{"url":"/paper/unified-pre-training-for-program","title":"Unified Pre-training for Program Understanding and Generation","date":"2021-03-10","arxiv_id":"2103.06333","repositories_listed":2,"syntology":null},{"url":"/paper/dobf-a-deobfuscation-pre-training-objective","title":"DOBF: A Deobfuscation Pre-Training Objective for Programming Languages","date":"2021-02-15","arxiv_id":"2102.07492","repositories_listed":2,"syntology":null},{"url":"/paper/simplifying-models-with-unlabeled-output-data","title":"Composed Fine-Tuning: Freezing Pre-Trained Denoising Autoencoders for Improved Generalization","date":"2020-06-29","arxiv_id":"2006.16205","repositories_listed":2,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/mutual-supervised-learning-for-sequential-to","title":"Mutual-Supervised Learning for Sequential-to-Parallel Code Translation","date":"2025-06-11","arxiv_id":"2506.11153","repositories_listed":1,"syntology":null},{"url":"/paper/timeseriesgym-a-scalable-benchmark-for-time","title":"TimeSeriesGym: A Scalable Benchmark for (Time Series) Machine Learning Engineering Agents","date":"2025-05-19","arxiv_id":"2505.13291","repositories_listed":1,"syntology":null},{"url":"/paper/crust-bench-a-comprehensive-benchmark-for-c","title":"CRUST-Bench: A Comprehensive Benchmark for C-to-safe-Rust Transpilation","date":"2025-04-21","arxiv_id":"2504.15254","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/nvagent-automated-data-visualization-from","title":"nvAgent: Automated Data Visualization from Natural Language via Collaborative Agent Workflow","date":"2025-02-07","arxiv_id":"2502.05036","repositories_listed":1,"syntology":null},{"url":"/paper/green-code-optimizing-energy-efficiency-in","title":"GREEN-CODE: Learning to Optimize Energy Efficiency in LLM-based Code Generation","date":"2025-01-19","arxiv_id":"2501.11006","repositories_listed":1,"syntology":null},{"url":"/paper/fortran2cpp-automating-fortran-to-c-migration","title":"Fortran2CPP: Automating Fortran-to-C++ Translation using LLMs via Multi-Turn Dialogue and Dual-Agent Integration","date":"2024-12-27","arxiv_id":"2412.19770","repositories_listed":1,"syntology":null},{"url":"/paper/transducer-tuning-efficient-model-adaptation","title":"Transducer Tuning: Efficient Model Adaptation for Software Tasks Using Code Property Graphs","date":"2024-12-18","arxiv_id":"2412.13467","repositories_listed":1,"syntology":null},{"url":"/paper/intertrans-leveraging-transitive-intermediate","title":"InterTrans: Leveraging Transitive Intermediate Translations to Enhance LLM-based Code Translation","date":"2024-11-01","arxiv_id":"2411.01063","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-models-for-code","title":"Leveraging Large Language Models for Code Translation and Software Development in Scientific Computing","date":"2024-10-31","arxiv_id":"2410.24119","repositories_listed":1,"syntology":null},{"url":"/paper/repository-level-compositional-code","title":"AlphaTrans: A Neuro-Symbolic Compositional Approach for Repository-Level Code Translation and Validation","date":"2024-10-31","arxiv_id":"2410.24117","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/rectifier-code-translation-with-corrector-via","title":"Rectifier: Code Translation with Corrector via LLMs","date":"2024-07-10","arxiv_id":"2407.07472","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","arxiv_id":"2406.20098","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":7}},{"url":"/paper/exploring-and-unleashing-the-power-of-large","title":"Exploring and Unleashing the Power of Large Language Models in Automated Code Translation","date":"2024-04-23","arxiv_id":"2404.14646","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/exploring-the-impact-of-the-output-format-on","title":"Exploring the Impact of the Output Format on the Evaluation of Large Language Models for Code Translation","date":"2024-03-25","arxiv_id":"2403.17214","repositories_listed":1,"syntology":null},{"url":"/paper/unlocking-the-power-of-large-language-models","title":"Unlocking the Power of Large Language Models for Entity Alignment","date":"2024-02-23","arxiv_id":"2402.15048","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/intervenor-prompt-the-coding-ability-of-large","title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","date":"2023-11-16","arxiv_id":"2311.09868","repositories_listed":1,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}