{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/pubmed-200k-rct-a-dataset-for-sequential","title":"PubMed 200k RCT: a Dataset for Sequential Sentence Classification in Medical Abstracts","arxiv_id":"1710.06071","date":"2017-10-17","proceeding":"IJCNLP 2017 11","authors":["Franck Dernoncourt","Ji Young Lee"],"abstract":"We present PubMed 200k RCT, a new dataset based on PubMed for sequential\nsentence classification. The dataset consists of approximately 200,000\nabstracts of randomized controlled trials, totaling 2.3 million sentences. Each\nsentence of each abstract is labeled with their role in the abstract using one\nof the following classes: background, objective, method, result, or conclusion.\nThe purpose of releasing this dataset is twofold. First, the majority of\ndatasets for sequential short-text classification (i.e., classification of\nshort texts that appear in sequences) are small: we hope that releasing a new\nlarge dataset will help develop more accurate algorithms for this task. Second,\nfrom an application perspective, researchers need better tools to efficiently\nskim through the literature. Automatically classifying each sentence in an\nabstract would help researchers read abstracts more efficiently, especially in\nfields where abstracts may be long, such as the medical field.","url_abs":"http://arxiv.org/abs/1710.06071v1","url_pdf":"http://arxiv.org/pdf/1710.06071v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/Franck-Dernoncourt/pubmed-rct","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/DCYN/Ramdomized-Clinical-Trail-Classification","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/YPCC/medical-data","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/joelweber97/Python3_TF_Certificate","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/mrdbourke/tensorflow-deep-learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/pranawmishra/SkimLit","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/thisiskhan/tensorflow-developer-certificate-machine-learning-kit","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/tomwalczak/PubMed-Abstract-Analyzer","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"pubmed-200k-rct-a-dataset-for-sequential","repo_url":"https://github.com/vishalrk1/SkimLit","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"sentence-classification","task_name":"Sentence Classification"},{"task_slug":"text-classification","task_name":"Text Classification"}],"methods":[],"datasets_introduced":[{"slug":"pubmed-rct","name":"PubMed RCT","full_name":"PubMed 200k RCT"}],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/1710.06071","atlas_url":"https://app.syntology.ai/?focus=1710.06071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.06071"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mrdbourke/tensorflow-deep-learning","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pranawmishra/SkimLit","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tomwalczak/PubMed-Abstract-Analyzer","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/vishalrk1/SkimLit","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/joelweber97/Python3_TF_Certificate","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/DCYN/Ramdomized-Clinical-Trail-Classification","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/thisiskhan/tensorflow-developer-certificate-machine-learning-kit","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Franck-Dernoncourt/pubmed-rct","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/YPCC/medical-data","reach":{"status":"ok"}}],"summary":{"ran_draft_wrong":1},"by_repo_kind":{"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"d31ee189c0563e3f","entry":"gather_last_relevant_hidden","repo":"vishalrk1/SkimLit","repo_kind":"listed","path":"Model.py","file_url":"https://github.com/vishalrk1/SkimLit/blob/HEAD/Model.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d31ee189c0563e3f"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}