{"url":"/dataset/code-lingua","name":"Code Lingua","full_name":null,"description_markdown":"Code Lingua is a benchmark to compare the ability of language models to understand what the code implements in the source language and translate the same semantics in the target language. It comprises 1,700 code samples across five programming languages, over 10,000 tests, 43,000 translated code snippets, 1,748 manually labeled bugs, and 1,365 bug-fix pairs.","description_withheld":null,"homepage":"https://github.com/codetlingua/codetlingua","introduced_date":"2023-08-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/understanding-the-effectiveness-of-large","title":"Lost in Translation: A Study of Bugs Introduced by Large Language Models while Translating Code","first_author":null,"url":null},"license":null,"modalities":[],"tasks":[],"languages":[],"variants":["Code Lingua"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}