{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/pre-training-meets-clustering-a-hybrid","title":"Pre-training Meets Clustering: A Hybrid Extractive Multi-document Summarization Model","arxiv_id":null,"date":"2023-05-25","proceeding":"International Conference on Hybrid Intelligent Systems 2023 5","authors":["Akanksha Karotia","Seba Susan"],"abstract":"In this era where a large amount of information has flooded the Internet, manual extraction and consumption of relevant information is very difficult and time-consuming. Therefore, an automated document summarization tool is necessary to excerpt important information from a set of documents that have similar or related subjects. Multi-document summarization allows retrieval of important and relevant content from multiple documents while minimizing redundancy. A multi-document text summarization system is developed in this study using an unsupervised extractive-based approach. The proposed model is a fusion of two learning paradigms: the T5 pre-trained transformer model and the K-Means clustering algorithm. We perform the experiments on the benchmark news article corpus Document Understanding Conference (DUC2004). The ROUGE evaluation metrics were used to estimate the performance of the proposed approach on the DUC2004. Outcomes validate that our proposed model shows greatly enhanced performance as compared to the existent unsupervised state-of-the-art approaches.","url_abs":"https://link.springer.com/chapter/10.1007/978-3-031-27409-1_48","url_pdf":"https://www.researchgate.net/publication/371032596_Pre-training_Meets_Clustering_A_Hybrid_Extractive_Multi-document_Summarization_Model/link/647405096fb1d1682b18b1d4/download","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"pre-training-meets-clustering-a-hybrid","repo_url":"https://github.com/Akankshakarotia/Pre-training-meets-Clustering-A-Hybrid-Extractive-Multi-Document-Summarization-Model","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"clustering","task_name":"Clustering"},{"task_slug":"document-summarization","task_name":"Document Summarization"},{"task_slug":"extractive-document-summarization","task_name":"Extractive Text Summarization"},{"task_slug":"multi-document-summarization","task_name":"Multi-Document Summarization"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"text-summarization","task_name":"Text Summarization"},{"task_slug":"unsupervised-text-summarization","task_name":"Unsupervised Text Summarization"},{"task_slug":"document-understanding","task_name":"document understanding"}],"methods":[{"method_slug":"adafactor","method_name":"Adafactor"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"glu","method_name":"Gated Linear Unit"},{"method_slug":"inverse-square-root-schedule","method_name":"Inverse Square Root Schedule"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"sentencepiece","method_name":"SentencePiece"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"t5","method_name":"T5"},{"method_slug":"k-means-clustering","method_name":"k-Means Clustering"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/extractive-text-summarization-on-duc-2004-1","task":"Extractive Text Summarization","dataset":"DUC 2004","model":"Pre-training-meets-Clustering-A-Hybrid-Extractive-Multi-Document-Summarization-Model","rank_in_archive_order":1,"of":1,"metrics":{"Test ROGUE-1":"34.013","Test ROGUE-2":"8.266"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}