{"url":"/dataset/bucc","name":"BUCC","full_name":"Building and Using Comparable Corpora","description_markdown":"The **BUCC** mining task is a shared task on parallel sentence extraction from two monolingual corpora with a subset of them assumed to be parallel, and that has been available since 2016. For each language pair, the shared task provides a monolingual corpus for each language and a gold mapping list containing true translation pairs. These pairs are the ground truth. The task is to construct a list of translation pairs from the monolingual corpora. The constructed list is compared to the ground truth, and evaluated in terms of the F1 measure.\r\n\r\nSource: [Language-agnostic BERT Sentence Embedding](https://arxiv.org/abs/2007.01852)\r\nImage Source: [https://comparable.limsi.fr/bucc2017/](https://comparable.limsi.fr/bucc2017/)","description_withheld":null,"homepage":"https://comparable.limsi.fr/bucc2017/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/overview-of-the-second-bucc-shared-task","title":"Overview of the Second BUCC Shared Task: Spotting Parallel Sentences in Comparable Corpora","first_author":"Pierre Zweigenbaum","url":null},"license":{"name":"Custom","url":"https://comparable.limsi.fr/bucc2017/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Cross-Lingual Bitext Mining","url":"/task/cross-lingual-bitext-mining","datasets_with_task":"/datasets/task/cross-lingual-bitext-mining"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"German","url":"/datasets/language/german"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["BUCC German-to-English","BUCC French-to-English","BUCC Russian-to-English","BUCC Chinese-to-English","BUCC"],"data_loaders":[],"num_papers_in_archive":42,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/cross-lingual-bitext-mining-on-bucc-french-to","task":"Cross-Lingual Bitext Mining","dataset_variant":"BUCC French-to-English","rows":3,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"Massively Multilingual Sentence Embeddings","paper":"/paper/massively-multilingual-sentence-embeddings","metrics":{"F1 score":"93.91"},"code_links":[{"title":"facebookresearch/LASER","url":"https://github.com/facebookresearch/LASER"},{"title":"Unbabel/COMET","url":"https://github.com/Unbabel/COMET"},{"title":"facebookresearch/vizseq","url":"https://github.com/facebookresearch/vizseq"},{"title":"yannvgn/laserembeddings","url":"https://github.com/yannvgn/laserembeddings"},{"title":"transducens/LASERtrain","url":"https://github.com/transducens/LASERtrain"},{"title":"jeongukjae/smaller-labse","url":"https://github.com/jeongukjae/smaller-labse"},{"title":"raymondhs/fairseq-laser","url":"https://github.com/raymondhs/fairseq-laser"},{"title":"jiamingkong/infoxlm_paddle","url":"https://github.com/jiamingkong/infoxlm_paddle"},{"title":"Tony4469/laser-agir","url":"https://github.com/Tony4469/laser-agir"},{"title":"kmkwon94/ainize-laser","url":"https://github.com/kmkwon94/ainize-laser"},{"title":"imamathcat/LASER_Dependencies","url":"https://github.com/imamathcat/LASER_Dependencies"},{"title":"LawrenceDuan/myLASER","url":"https://github.com/LawrenceDuan/myLASER"},{"title":"prabhakar267/LASER-improved","url":"https://github.com/prabhakar267/LASER-improved"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-bitext-mining-on-bucc-german-to","task":"Cross-Lingual Bitext Mining","dataset_variant":"BUCC German-to-English","rows":3,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"Massively Multilingual Sentence Embeddings","paper":"/paper/massively-multilingual-sentence-embeddings","metrics":{"F1 score":"96.19"},"code_links":[{"title":"facebookresearch/LASER","url":"https://github.com/facebookresearch/LASER"},{"title":"Unbabel/COMET","url":"https://github.com/Unbabel/COMET"},{"title":"facebookresearch/vizseq","url":"https://github.com/facebookresearch/vizseq"},{"title":"yannvgn/laserembeddings","url":"https://github.com/yannvgn/laserembeddings"},{"title":"transducens/LASERtrain","url":"https://github.com/transducens/LASERtrain"},{"title":"jeongukjae/smaller-labse","url":"https://github.com/jeongukjae/smaller-labse"},{"title":"raymondhs/fairseq-laser","url":"https://github.com/raymondhs/fairseq-laser"},{"title":"jiamingkong/infoxlm_paddle","url":"https://github.com/jiamingkong/infoxlm_paddle"},{"title":"Tony4469/laser-agir","url":"https://github.com/Tony4469/laser-agir"},{"title":"kmkwon94/ainize-laser","url":"https://github.com/kmkwon94/ainize-laser"},{"title":"imamathcat/LASER_Dependencies","url":"https://github.com/imamathcat/LASER_Dependencies"},{"title":"LawrenceDuan/myLASER","url":"https://github.com/LawrenceDuan/myLASER"},{"title":"prabhakar267/LASER-improved","url":"https://github.com/prabhakar267/LASER-improved"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-bitext-mining-on-bucc-chinese","task":"Cross-Lingual Bitext Mining","dataset_variant":"BUCC Chinese-to-English","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"Massively Multilingual Sentence Embeddings","paper":"/paper/massively-multilingual-sentence-embeddings","metrics":{"F1 score":"92.27"},"code_links":[{"title":"facebookresearch/LASER","url":"https://github.com/facebookresearch/LASER"},{"title":"Unbabel/COMET","url":"https://github.com/Unbabel/COMET"},{"title":"facebookresearch/vizseq","url":"https://github.com/facebookresearch/vizseq"},{"title":"yannvgn/laserembeddings","url":"https://github.com/yannvgn/laserembeddings"},{"title":"transducens/LASERtrain","url":"https://github.com/transducens/LASERtrain"},{"title":"jeongukjae/smaller-labse","url":"https://github.com/jeongukjae/smaller-labse"},{"title":"raymondhs/fairseq-laser","url":"https://github.com/raymondhs/fairseq-laser"},{"title":"jiamingkong/infoxlm_paddle","url":"https://github.com/jiamingkong/infoxlm_paddle"},{"title":"Tony4469/laser-agir","url":"https://github.com/Tony4469/laser-agir"},{"title":"kmkwon94/ainize-laser","url":"https://github.com/kmkwon94/ainize-laser"},{"title":"imamathcat/LASER_Dependencies","url":"https://github.com/imamathcat/LASER_Dependencies"},{"title":"LawrenceDuan/myLASER","url":"https://github.com/LawrenceDuan/myLASER"},{"title":"prabhakar267/LASER-improved","url":"https://github.com/prabhakar267/LASER-improved"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-bitext-mining-on-bucc-russian","task":"Cross-Lingual Bitext Mining","dataset_variant":"BUCC Russian-to-English","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"Massively Multilingual Sentence Embeddings","paper":"/paper/massively-multilingual-sentence-embeddings","metrics":{"F1 score":"93.3"},"code_links":[{"title":"facebookresearch/LASER","url":"https://github.com/facebookresearch/LASER"},{"title":"Unbabel/COMET","url":"https://github.com/Unbabel/COMET"},{"title":"facebookresearch/vizseq","url":"https://github.com/facebookresearch/vizseq"},{"title":"yannvgn/laserembeddings","url":"https://github.com/yannvgn/laserembeddings"},{"title":"transducens/LASERtrain","url":"https://github.com/transducens/LASERtrain"},{"title":"jeongukjae/smaller-labse","url":"https://github.com/jeongukjae/smaller-labse"},{"title":"raymondhs/fairseq-laser","url":"https://github.com/raymondhs/fairseq-laser"},{"title":"jiamingkong/infoxlm_paddle","url":"https://github.com/jiamingkong/infoxlm_paddle"},{"title":"Tony4469/laser-agir","url":"https://github.com/Tony4469/laser-agir"},{"title":"kmkwon94/ainize-laser","url":"https://github.com/kmkwon94/ainize-laser"},{"title":"imamathcat/LASER_Dependencies","url":"https://github.com/imamathcat/LASER_Dependencies"},{"title":"LawrenceDuan/myLASER","url":"https://github.com/LawrenceDuan/myLASER"},{"title":"prabhakar267/LASER-improved","url":"https://github.com/prabhakar267/LASER-improved"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/massively-multilingual-sentence-embeddings","title":"Massively Multilingual Sentence Embeddings for Zero-Shot Cross-Lingual Transfer and Beyond","date":"2018-12-26","rows_on_this_dataset":4,"code_links":13,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":10,"samples_ran":4,"samples_unverified":6,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/margin-based-parallel-corpus-mining-with","title":"Margin-based Parallel Corpus Mining with Multilingual Sentence Embeddings","date":"2018-11-03","rows_on_this_dataset":2,"code_links":9,"syntology":null},{"paper":"/paper/improving-neural-machine-translation-models","title":"Improving Neural Machine Translation Models with Monolingual Data","date":"2015-11-20","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":2,"samples_harvested":16,"samples_ran":10,"samples_unverified":6,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}