{"url":"/dataset/tdcommons","name":"tdcommons","full_name":"Therapeutics Data Commons","description_markdown":"Therapeutics Data Commons is an open-science initiative with AI/ML-ready datasets and AI/ML tasks for therapeutics, spanning the discovery and development of safe and effective medicines. TDC provides an ecosystem of tools, libraries, leaderboards, and community resources, including data functions, strategies for systematic model evaluation, meaningful data splits, data processors, and molecule generation oracles. All resources are integrated via an open Python library.","description_withheld":null,"homepage":"https://tdcommons.ai","introduced_date":"2021-02-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/therapeutics-data-commons-machine-learning","title":"Therapeutics Data Commons: Machine Learning Datasets and Tasks for Drug Discovery and Development","first_author":"Kexin Huang","url":null},"license":{"name":"MIT License","url":"https://github.com/mims-harvard/TDC/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"TDC ADMET Benchmarking Group","url":"/task/tdc-admet-benchmarking-group","datasets_with_task":"/datasets/task/tdc-admet-benchmarking-group"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["tdcommons"],"data_loaders":[{"repo":"https://github.com/mims-harvard/TDC","url":"https://github.com/mims-harvard/TDC","frameworks":[]}],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/tdc-admet-benchmarking-group-on-tdcommons","task":"TDC ADMET Benchmarking Group","dataset_variant":"tdcommons","rows":12,"metrics":["TDC.Caco2_Wang","TDC.HIA_Hou","TDC.Pgp_Broccatelli","TDC.Bioavailability_Ma","TDC.Lipophilicity_AstraZeneca","TDC.Solubility_AqSolDB","TDC.BBB_Martins","TDC.PPBR_AZ","TDC.VDss_Lombardo","TDC.CYP2D6_Inhibition_Veith","TDC.CYP3A4_Inhibition_Veith","TDC.CYP2C9_Inhibition_Veith","TDC.CYP2D6_Substrate_CarbonMangels","TDC.CYP3A4_Substrate_CarbonMangels","TDC.CYP2C9_Substrate_CarbonMangels","TDC.Half_Life_Obach","TDC.Clearance_Microsome_AZ","TDC.Clearance_Hepatocyte_AZ","TDC.hERG","TDC.AMES","TDC.DILI","TDC.LD50_Zhu"],"first_row_in_archive_order":{"model":"MapLight","paper":"/paper/admet-property-prediction-through","metrics":{"TDC.Caco2_Wang":"0.276"},"code_links":[{"title":"maplightrx/maplight-tdc","url":"https://github.com/maplightrx/maplight-tdc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/deepmol-an-automated-machine-and-deep","title":"DeepMol: An Automated Machine and Deep Learning Framework for Computational Chemistr","date":"2024-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/admet-property-prediction-through","title":"ADMET property prediction through combinations of molecular fingerprints","date":"2023-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":5,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/accurate-admet-prediction-with-xgboost","title":"Accurate ADMET Prediction with XGBoost","date":"2022-04-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/therapeutics-data-commons-machine-learning","title":"Therapeutics Data Commons: Machine Learning Datasets and Tasks for Drug Discovery and Development","date":"2021-02-18","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}