{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/when-smiles-have-language-drug-classification","title":"When SMILES have Language: Drug Classification using Text Classification Methods on Drug SMILES Strings","arxiv_id":"2403.12984","date":"2024-03-03","proceeding":null,"authors":["Azmine Toushik Wasi","Šerbetar Karlo","Raima Islam","Taki Hasan Rafi","Dong-Kyu Chae"],"abstract":"Complex chemical structures, like drugs, are usually defined by SMILES strings as a sequence of molecules and bonds. These SMILES strings are used in different complex machine learning-based drug-related research and representation works. Escaping from complex representation, in this work, we pose a single question: What if we treat drug SMILES as conventional sentences and engage in text classification for drug classification? Our experiments affirm the possibility with very competitive scores. The study explores the notion of viewing each atom and bond as sentence components, employing basic NLP methods to categorize drug types, proving that complex problems can also be solved with simpler perspectives. The data and code are available here: https://github.com/azminewasi/Drug-Classification-NLP.","url_abs":"https://arxiv.org/abs/2403.12984v2","url_pdf":"https://arxiv.org/pdf/2403.12984v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"when-smiles-have-language-drug-classification","repo_url":"https://github.com/azminewasi/Drug-Classification-NLP","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"classification-1","task_name":"Classification"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"text-classification","task_name":"Text Classification"},{"task_slug":"text-classification-1","task_name":"text-classification"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2403.12984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12984"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/azminewasi/Drug-Classification-NLP","reach":null}],"summary":{"ran_draft_wrong":4,"ran_honours":1},"by_repo_kind":{"official":{"samples":5,"ran":5,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"65a5427c773a7ab1","entry":"make_n_gram_mapping","repo":"azminewasi/Drug-Classification-NLP","repo_kind":"official","path":"Model/data/dataset.py","file_url":"https://github.com/azminewasi/Drug-Classification-NLP/blob/HEAD/Model/data/dataset.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"65a5427c773a7ab1"}},{"code_sha256_prefix":"63e4b2232457023a","entry":"read_csv","repo":"azminewasi/Drug-Classification-NLP","repo_kind":"official","path":"Model/data/dataset.py","file_url":"https://github.com/azminewasi/Drug-Classification-NLP/blob/HEAD/Model/data/dataset.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"63e4b2232457023a"}},{"code_sha256_prefix":"5a315c1d56004725","entry":"split_data","repo":"azminewasi/Drug-Classification-NLP","repo_kind":"official","path":"Model/data/dataset.py","file_url":"https://github.com/azminewasi/Drug-Classification-NLP/blob/HEAD/Model/data/dataset.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5a315c1d56004725"}},{"code_sha256_prefix":"5a0b9a4073d697c0","entry":"test","repo":"azminewasi/Drug-Classification-NLP","repo_kind":"official","path":"Model/train-ngram.py","file_url":"https://github.com/azminewasi/Drug-Classification-NLP/blob/HEAD/Model/train-ngram.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5a0b9a4073d697c0"}},{"code_sha256_prefix":"ed294c3995b58c42","entry":"train","repo":"azminewasi/Drug-Classification-NLP","repo_kind":"official","path":"Model/train-ngram.py","file_url":"https://github.com/azminewasi/Drug-Classification-NLP/blob/HEAD/Model/train-ngram.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ed294c3995b58c42"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}