{"url":"/dataset/phomt","name":"PhoMT","full_name":null,"description_markdown":"**PhoMT** is a high-quality and large-scale Vietnamese-English parallel dataset of 3.02M sentence pairs for machine translation.","description_withheld":null,"homepage":"https://github.com/VinAIResearch/PhoMT","introduced_date":"2021-10-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/phomt-a-high-quality-and-large-scale","title":"PhoMT: A High-Quality and Large-Scale Benchmark Dataset for Vietnamese-English Machine Translation","first_author":"Long Doan","url":null},"license":{"name":"Custom","url":"https://github.com/VinAIResearch/PhoMT#copyright-c-2021-vinai-research"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"},{"name":"Translation","url":"/task/translation","datasets_with_task":"/datasets/task/translation"}],"languages":[{"name":"Vietnamese","url":"/datasets/language/vietnamese"}],"variants":["PhoMT"],"data_loaders":[{"repo":"https://github.com/imsteps/projec","url":"https://github.com/imsteps/projec","frameworks":["tf","pytorch"]}],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/translation-on-phomt","task":"Translation","dataset_variant":"PhoMT","rows":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"ALMA-Gemma-7B-IT-ST","paper":"/paper/advanced-language-model-based-translator-for","metrics":{"BLEU":"56.21"},"code_links":[{"title":"doctranslate-io/viet-translation-llm","url":"https://github.com/doctranslate-io/viet-translation-llm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/advanced-language-model-based-translator-for","title":"Advanced Language Model-based Translator for English-Vietnamese Translation","date":"2024-05-27","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}