{"url":"/dataset/atc-smiles","name":"ATC-SMILES","full_name":null,"description_markdown":"The benchmark ATC-SMILES is built for ATC classification. ATC-SMILES consists of 4545 compounds/drugs and their SMILES sequences. The benchmark is with the maximum coverage (81.34%) of KEGG dataset which contains all 5588 known drugs/compounds used for ATC analysis. Prior to this benchmark, the most widely adopted one is Chen-2012 which covers 3883 (69.49%) drugs in KEGG and is mainly used for generating inter-drug correlations (e.g. STITCH). The two benchmarks are compared in Table 1. ATC-SMILES is designed to be inclusive to Chen-2012, but there are 2.16% misalignment due to the mismatching of drug IDs that we will explain soon. ATC-SMILES can be extended with new drugs much easier than previous benchmarks as long as the SMILES sequences are available. Trails/experiments are not a must.","description_withheld":null,"homepage":"https://github.com/lookwei/ATC_CNN","introduced_date":"2022-08-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/identifying-the-kind-behind-smiles-anatomical","title":"Identifying the kind behind SMILES—anatomical therapeutic chemical classification using structure-only representations","first_author":"Yi Cao","url":null},"license":null,"modalities":[],"tasks":[{"name":"Drug ATC Classification","url":"/task/drug-atc-classification","datasets_with_task":"/datasets/task/drug-atc-classification"}],"languages":[],"variants":["ATC-SMILES"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/drug-atc-classification-on-atc-smiles","task":"Drug ATC Classification","dataset_variant":"ATC-SMILES","rows":2,"metrics":["Absolute False","Absolute True","Accuracy","Aiming","Coverage"],"first_row_in_archive_order":{"model":"GraphATC","paper":"/paper/graphatc-advancing-multilevel-and-multi-label","metrics":{"Absolute False":"0.0068","Absolute True":"0.9397","Accuracy":"0.9542","Aiming":"0.9608","Coverage":"0.9609"},"code_links":[{"title":"lookwei/GraphATC","url":"https://github.com/lookwei/GraphATC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/graphatc-advancing-multilevel-and-multi-label","title":"GraphATC: advancing multilevel and multi-label anatomical therapeutic chemical classification via atom-level graph learning","date":"2025-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/identifying-the-kind-behind-smiles-anatomical","title":"Identifying the kind behind SMILES—anatomical therapeutic chemical classification using structure-only representations","date":"2022-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}