{"url":"/dataset/chebi-20","name":"ChEBI-20","full_name":null,"description_markdown":"Dataset contains 33,010 molecule-description pairs split into 80\\%/10\\%/10\\% train/val/test splits. The goal of the task is to retrieve the relevant molecule for a natural language description. It is defined as follows:\r\n\r\nTo push the boundaries of multimodal models, we present a new IR task: \\textbf{Text2Mol}.\r\n\r\nGiven a text query and list of molecules without any reference textual information (represented, for example, as SMILES strings, graphs, or other equivalent representations) retrieve the molecule corresponding to the query. From a text description of a molecule, the model must incorporate the information in the description into a semantic representation which can be used to directly retrieve the molecule. This requires the integration of two very different types of information: the structured knowledge represented by text and the chemical properties present in molecular graphs. We assume there is only one correct (relevant) molecule for each description, so we consider two measures for this task: Hits@1 and mean reciprocal rank (MRR). \r\n\r\n80\\% of the data is used for training. Retrieval is done against the entire corpus of molecules (train, val, test).","description_withheld":null,"homepage":"https://github.com/cnedwards/text2mol/tree/master/data","introduced_date":"2021-11-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/text2mol-cross-modal-molecule-retrieval-with","title":"Text2Mol: Cross-Modal Molecule Retrieval with Natural Language Queries","first_author":"Carl Edwards","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Graphs","url":"/datasets/modality/graphs"},{"name":"Biomedical","url":"/datasets/modality/biomedical"}],"tasks":[{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Molecule Captioning","url":"/task/molecule-captioning","datasets_with_task":"/datasets/task/molecule-captioning"},{"name":"Text-based de novo Molecule Generation","url":"/task/text-based-de-novo-molecule-generation","datasets_with_task":"/datasets/task/text-based-de-novo-molecule-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ChEBI-20"],"data_loaders":[],"num_papers_in_archive":43,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/molecule-captioning-on-chebi-20","task":"Molecule Captioning","dataset_variant":"ChEBI-20","rows":33,"metrics":["BLEU-2","BLEU-4","METEOR","ROUGE-1","ROUGE-2","ROUGE-L","Text2Mol"],"first_row_in_archive_order":{"model":"Mol-LLM (Mistral-Instruct-v0.2)","paper":null,"metrics":{"BLEU-2":"73.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-based-de-novo-molecule-generation-on","task":"Text-based de novo Molecule Generation","dataset_variant":"ChEBI-20","rows":20,"metrics":["BLEU","Exact Match","Frechet ChemNet Distance (FCD)","Levenshtein","MACCS FTS","Morgan FTS","RDK FTS","Text2Mol","Validity","Parameter Count"],"first_row_in_archive_order":{"model":"LDMol","paper":"/paper/ldmol-text-conditioned-molecule-diffusion","metrics":{"BLEU":"92.6","Exact Match":"53.3","Frechet ChemNet Distance (FCD)":"0.20","Levenshtein":"6.750","MACCS FTS":"97.3","Morgan FTS":"93.1","RDK FTS":"95.0","Validity":"94.1"},"code_links":[{"title":"jinhojsk515/ldmol","url":"https://github.com/jinhojsk515/ldmol"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-modal-retrieval-on-chebi-20","task":"Cross-Modal Retrieval","dataset_variant":"ChEBI-20","rows":9,"metrics":["Hits@1","Hits@10","Mean Rank","Test MRR"],"first_row_in_archive_order":{"model":"CLASS (ORMA)","paper":"/paper/class-enhancing-cross-modal-text-molecule","metrics":{"Hits@1":"67.4","Hits@10":"93.4","Mean Rank":"17.82","Test MRR":"77.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-chebi-20","task":"Image Captioning","dataset_variant":"ChEBI-20","rows":1,"metrics":["BLEU","Exact","Levenshtein","MACCS FTS","Morgan FTS","RDK FTS","Validity"],"first_row_in_archive_order":{"model":"GIT-Mol","paper":"/paper/git-mol-a-multi-modal-large-language-model","metrics":{"BLEU":"0.924","Exact":"0.461","Levenshtein":"6.575","MACCS FTS":"0.962","Morgan FTS":"0.894","RDK FTS":"0.906","Validity":"0.899"},"code_links":[{"title":"ai-hpc-research-team/git-mol","url":"https://github.com/ai-hpc-research-team/git-mol"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/xmolcap-advancing-molecular-captioning","title":"XMolCap: Advancing Molecular Captioning through Multimodal Fusion and Explainable Graph Neural Networks","date":"2025-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/class-enhancing-cross-modal-text-molecule","title":"CLASS: Enhancing Cross-Modal Text-Molecule Retrieval Performance and Training Efficiency","date":"2025-02-17","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/automatic-annotation-augmentation-boosts","title":"Automatic Annotation Augmentation Boosts Translation between Molecules and Natural Language","date":"2025-02-10","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/mol-llm-generalist-molecular-llm-with","title":"Mol-LLM: Multimodal Generalist Molecular LLM with Improved Graph Utilization","date":"2025-02-05","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/property-enhanced-instruction-tuning-for","title":"Property Enhanced Instruction Tuning for Multi-task Molecule Generation with Large Language Models","date":"2024-12-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/molreflect-towards-fine-grained-in-context","title":"MolReFlect: Towards Fine-grained In-Context Alignment between Molecules and Texts","date":"2024-11-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/exploring-optimal-transport-based-multi","title":"Exploring Optimal Transport-Based Multi-Grained Alignments for Text-Molecule Retrieval","date":"2024-11-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/towards-cross-modal-text-molecule-retrieval","title":"Towards Cross-Modal Text-Molecule Retrieval with Better Modality Alignment","date":"2024-10-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mol2lang-vlm-vision-and-text-guided","title":"Mol2Lang-VLM: Vision- and Text-Guided Generative Pre-trained Language Models for Advancing Molecule Captioning through Multimodal Fusion","date":"2024-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-sketched-output-kernel-regression-for","title":"Deep Sketched Output Kernel Regression for Structured Prediction","date":"2024-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ldmol-text-conditioned-molecule-diffusion","title":"LDMol: Text-to-Molecule Diffusion Model with Structurally Informative Latent Space","date":"2024-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biot5-towards-generalized-biological","title":"BioT5+: Towards Generalized Biological Understanding with IUPAC Integration and Multi-task Tuning","date":"2024-02-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/text-guided-molecule-generation-with","title":"Text-Guided Molecule Generation with Diffusion Language Model","date":"2024-02-20","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":14,"samples_unverified":7,"pointer_only_for_licence":21,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/instructmol-multi-modal-integration-for","title":"InstructMol: Multi-Modal Integration for Building a Versatile and Reliable Molecular Assistant in Drug Discovery","date":"2023-11-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/molca-molecular-graph-language-modeling-with","title":"MolCA: Molecular Graph-Language Modeling with Cross-Modal Projector and Uni-Modal Adapter","date":"2023-10-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":8,"samples_unverified":8,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biot5-enriching-cross-modal-integration-in","title":"BioT5: Enriching Cross-modal Integration in Biology with Chemical Knowledge and Natural Language Associations","date":"2023-10-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/git-mol-a-multi-modal-large-language-model","title":"GIT-Mol: A Multi-modal Large Language Model for Molecular Science with Graph, Image, and Text","date":"2023-08-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/empowering-molecule-discovery-for-molecule","title":"Empowering Molecule Discovery for Molecule-Caption Translation with Large Language Models: A ChatGPT Perspective","date":"2023-06-11","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/molfm-a-multimodal-molecular-foundation-model","title":"MolFM: A Multimodal Molecular Foundation Model","date":"2023-06-06","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/molxpt-wrapping-molecules-with-text-for","title":"MolXPT: Wrapping Molecules with Text for Generative Pre-training","date":"2023-05-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/adversarial-modality-alignment-network-for","title":"Adversarial Modality Alignment Network for Cross-Modal Molecule Retrieval","date":"2023-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifying-molecular-and-textual","title":"Unifying Molecular and Textual Representations via Multi-task Language Modelling","date":"2023-01-29","rows_on_this_dataset":8,"code_links":1,"syntology":null},{"paper":"/paper/a-molecular-multimodal-foundation-model","title":"A Molecular Multimodal Foundation Model Associating Molecule Graphs with Natural Language","date":"2022-09-12","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-based-molecular-representation-learning","title":"Graph-based Molecular Representation Learning","date":"2022-07-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/translation-between-molecules-and-natural","title":"Translation between Molecules and Natural Language","date":"2022-04-25","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text2mol-cross-modal-molecule-retrieval-with","title":"Text2Mol: Cross-Modal Molecule Retrieval with Natural Language Queries","date":"2021-11-01","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":64,"samples_ran":35,"samples_unverified":29,"pointer_only_for_licence":40,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}