{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/debatesum-a-large-scale-argument-mining-and","title":"DebateSum: A large-scale argument mining and summarization dataset","arxiv_id":"2011.07251","date":"2020-11-14","proceeding":"COLING (ArgMining) 2020 12","authors":["Allen Roush","Arvind Balaji"],"abstract":"Prior work in Argument Mining frequently alludes to its potential applications in automatic debating systems. Despite this focus, almost no datasets or models exist which apply natural language processing techniques to problems found within competitive formal debate. To remedy this, we present the DebateSum dataset. DebateSum consists of 187,386 unique pieces of evidence with corresponding argument and extractive summaries. DebateSum was made using data compiled by competitors within the National Speech and Debate Association over a 7-year period. We train several transformer summarization models to benchmark summarization performance on DebateSum. We also introduce a set of fasttext word-vectors trained on DebateSum called debate2vec. Finally, we present a search engine for this dataset which is utilized extensively by members of the National Speech and Debate Association today. The DebateSum search engine is available to the public here: http://www.debate.cards","url_abs":"https://arxiv.org/abs/2011.07251v1","url_pdf":"https://arxiv.org/pdf/2011.07251v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"debatesum-a-large-scale-argument-mining-and","repo_url":"https://github.com/Hellisotherpeople/DebateSum","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"unanswered"}},{"paper_slug":"debatesum-a-large-scale-argument-mining-and","repo_url":"https://github.com/Hellisotherpeople/debate2vec","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"debatesum-a-large-scale-argument-mining-and","repo_url":"https://github.com/arvind-balaji/debate-cards","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"abstractive-text-summarization","task_name":"Abstractive Text Summarization"},{"task_slug":"argument-mining","task_name":"Argument Mining"},{"task_slug":"document-summarization","task_name":"Document Summarization"},{"task_slug":"extractive-document-summarization","task_name":"Extractive Text Summarization"},{"task_slug":"information-retrieval","task_name":"Information Retrieval"},{"task_slug":"query-based-extractive-summarization","task_name":"Query-Based Extractive Summarization"},{"task_slug":"text-summarization","task_name":"Text Summarization"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"adamw","method_name":"AdamW"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"cosine-annealing","method_name":"Cosine Annealing"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"discriminative-fine-tuning","method_name":"Discriminative Fine-Tuning"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"gpt-2","method_name":"GPT-2"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-cosine-annealing","method_name":"Linear Warmup With Cosine Annealing"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"longformer","method_name":"Longformer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"},{"method_slug":"fasttext","method_name":"fastText"}],"datasets_introduced":[{"slug":"debatesum","name":"DebateSum","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/extractive-document-summarization-on","task":"Extractive Text Summarization","dataset":"DebateSum","model":"Longformer-Base","rank_in_archive_order":1,"of":3,"metrics":{"ROUGE-L":"57.21"},"uses_additional_data":false},{"leaderboard":"/sota/extractive-document-summarization-on","task":"Extractive Text Summarization","dataset":"DebateSum","model":"GPT2-Medium","rank_in_archive_order":2,"of":3,"metrics":{"ROUGE-L":"53.23"},"uses_additional_data":false},{"leaderboard":"/sota/extractive-document-summarization-on","task":"Extractive Text Summarization","dataset":"DebateSum","model":"BERT-Large","rank_in_archive_order":3,"of":3,"metrics":{"ROUGE-L":"49.98"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2011.07251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07251"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/arvind-balaji/debate-cards","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Hellisotherpeople/debate2vec","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Hellisotherpeople/DebateSum","reach":{"status":"unanswered"}}],"summary":{"ran_draft_wrong":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"a130c85d44825679","entry":"parse_string","repo":"Hellisotherpeople/debate2vec","repo_kind":"official","path":"card_clustering.py","file_url":"https://github.com/Hellisotherpeople/debate2vec/blob/HEAD/card_clustering.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"mcp_get_code":{"code_sha256":"a130c85d44825679"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}