{"url":"/dataset/l-m-24","name":"L+M-24","full_name":null,"description_markdown":"Language-molecule models have emerged as an exciting direction for molecular discovery and understanding. However, training these models is challenging due to the scarcity of molecule-language pair datasets. At this point, datasets have been released which are 1) small and scraped from existing databases, 2) large but noisy and constructed by performing entity linking on the scientific literature, and 3) built by converting property prediction datasets to natural language using templates. In this document, we detail the L+M-24 dataset, which has been created for the Language + Molecules Workshop shared task at ACL 2024. In particular, L+M-24 is designed to focus on three key benefits of natural language in molecule design: compositionality, functionality, and abstraction","description_withheld":null,"homepage":"https://github.com/language-plus-molecules/LPM-24-Dataset","introduced_date":"2024-02-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/textit-l-m-24-building-a-dataset-for-language","title":"L+M-24: Building a Dataset for Language + Molecules @ ACL 2024","first_author":"Carl Edwards","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Graphs","url":"/datasets/modality/graphs"},{"name":"Biomedical","url":"/datasets/modality/biomedical"}],"tasks":[{"name":"Molecule Captioning","url":"/task/molecule-captioning","datasets_with_task":"/datasets/task/molecule-captioning"},{"name":"Text-based de novo Molecule Generation","url":"/task/text-based-de-novo-molecule-generation","datasets_with_task":"/datasets/task/text-based-de-novo-molecule-generation"},{"name":"Caption Generation","url":"/task/caption-generation","datasets_with_task":"/datasets/task/caption-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["L+M-24"],"data_loaders":[{"repo":"https://github.com/language-plus-molecules/lpm-24-dataset","url":"https://github.com/language-plus-molecules/lpm-24-dataset","frameworks":[]}],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/molecule-captioning-on-l-m-24","task":"Molecule Captioning","dataset_variant":"L+M-24","rows":6,"metrics":["BLEU-2","BLEU-4","ROUGE-1","ROUGE-2","ROUGE-L","METEOR"],"first_row_in_archive_order":{"model":"Mol2Lang-VLM","paper":"/paper/mol2lang-vlm-vision-and-text-guided","metrics":{"BLEU-2":"77.7","BLEU-4":"56.3","METEOR":"74.1","ROUGE-1":"78.6","ROUGE-2":"59.1","ROUGE-L":"56.5"},"code_links":[{"title":"nhattruongpham/mol-lang-bridge","url":"https://github.com/nhattruongpham/mol-lang-bridge"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/xmolcap-advancing-molecular-captioning","title":"XMolCap: Advancing Molecular Captioning through Multimodal Fusion and Explainable Graph Neural Networks","date":"2025-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mol2lang-vlm-vision-and-text-guided","title":"Mol2Lang-VLM: Vision- and Text-Guided Generative Pre-trained Language Models for Advancing Molecule Captioning through Multimodal Fusion","date":"2024-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/translation-between-molecules-and-natural","title":"Translation between Molecules and Natural Language","date":"2022-04-25","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}