{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/clustering-urdu-news-using-headlines","title":"Clustering Urdu News Using Headlines","arxiv_id":null,"date":"2015-09-27","proceeding":"23 2015 9","authors":["Samia Khaliq","Waheed Iqbal","Faisal Bukhari","Kamran Malik"],"abstract":"This paper that proposes and evaluates a new algorithm to automatically cluster Urdu news from different news agencies. The task is challenging because there are no language processing libraries for the Urdu language. The authors' experimental dataset consists of news from famous Pakistani media houses, including Jang, BBC Urdu, Express, UrduPoint, and Voice of America Urdu (VOA). The proposed algorithm only uses headlines to cluster the news. The authors argue that news headlines provide a concise summary of the news, which motivates them to use it instead of using the entire news story. Their experimental evaluation shows micro and macro averages for precision of 0.45 and 0.48 respectively for identifying similar news using headlines.","url_abs":"https://www.cle.org.pk/clt16/papers/Clustering%20Urdu%20News%20Using%20Headlines.pdf","url_pdf":"https://www.cle.org.pk/clt16/papers/Clustering%20Urdu%20News%20Using%20Headlines.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"clustering-urdu-news-using-headlines","repo_url":"https://github.com/SyedMuhammadFaheem/Urdu-News-Clustering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"clustering","task_name":"Clustering"},{"task_slug":"information-retrieval","task_name":"Information Retrieval"},{"task_slug":"text-clustering","task_name":"Text Clustering"}],"methods":[],"datasets_introduced":[{"slug":"urdu-news-headlines-dataset","name":"Urdu News Headlines Dataset","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/text-clustering-on-urdu-news-headlines","task":"Text Clustering","dataset":"Urdu News Headlines Dataset","model":"Vector Space Model","rank_in_archive_order":1,"of":1,"metrics":{"Related Headlines":"85"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}