{"url":"/task/spam-detection","name":"Spam detection","slug":"spam-detection","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":117,"papers_with_code":38,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/spam-detection-on-context-specific-spam","slug":"spam-detection-on-context-specific-spam","dataset":"Traditional and Context-specific Spam Twitter","dataset_url":"/dataset/context-specific-spam","rows_in_archive":1,"metrics":["Avg F1"],"first_row_in_archive_order":{"model":"BERT","paper_title":"Traditional and context-specific spam detection in low resource settings","paper_url":"/paper/traditional-and-context-specific-spam","paper_date":"2022-06-09","arxiv_id":null,"code_links":[{"title":"GU-DataLab/context-spam","url":"https://github.com/GU-DataLab/context-spam"}],"syntology":null}}],"datasets":[{"url":"/dataset/vispamreviews","name":"ViSpamReviews","full_name":"Vietnamese Spam Reviews Detection","num_papers_in_archive":3},{"url":"/dataset/context-specific-spam","name":"Traditional and Context-specific Spam Twitter","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/context-specific-spam-detection","name":"Context-specific Spam Detection"},{"url":"/task/traditional-spam-detection","name":"Traditional Spam Detection"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":38,"tagged_in_all":117,"items":[{"url":"/paper/gad-nr-graph-anomaly-detection-via","title":"GAD-NR: Graph Anomaly Detection via Neighborhood Reconstruction","date":"2023-06-02","arxiv_id":"2306.01951","repositories_listed":2,"syntology":null},{"url":"/paper/grouped-pointwise-convolutions-reduce","title":"Grouped Pointwise Convolutions Reduce Parameters in Convolutional Neural Networks","date":"2022-06-30","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/deep-convolutional-forest-a-dynamic-deep","title":"Deep convolutional forest: a dynamic deep ensemble approach for spam detection in text","date":"2021-10-10","arxiv_id":"2110.15718","repositories_listed":2,"syntology":null},{"url":"/paper/weight-poisoning-attacks-on-pre-trained","title":"Weight Poisoning Attacks on Pre-trained Models","date":"2020-04-14","arxiv_id":"2004.06660","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/training-robust-tree-ensembles-for-security","title":"Cost-Aware Robust Tree Ensembles for Security Applications","date":"2019-12-03","arxiv_id":"1912.01149","repositories_listed":2,"syntology":null},{"url":"/paper/gans-for-semi-supervised-opinion-spam","title":"GANs for Semi-Supervised Opinion Spam Detection","date":"2019-03-19","arxiv_id":"1903.08289","repositories_listed":2,"syntology":null},{"url":"/paper/stronger-data-poisoning-attacks-break-data","title":"Stronger Data Poisoning Attacks Break Data Sanitization Defenses","date":"2018-11-02","arxiv_id":"1811.00741","repositories_listed":2,"syntology":null},{"url":"/paper/evading-classifiers-in-discrete-domains-with","title":"Evading classifiers in discrete domains with provable optimality guarantees","date":"2018-10-25","arxiv_id":"1810.10939","repositories_listed":2,"syntology":null},{"url":"/paper/black-box-generation-of-adversarial-text","title":"Black-box Generation of Adversarial Text Sequences to Evade Deep Learning Classifiers","date":"2018-01-13","arxiv_id":"1801.04354","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/cyber-security-data-science-machine-learning","title":"Cyber Security Data Science: Machine Learning Methods and their Performance on Imbalanced Datasets","date":"2025-05-07","arxiv_id":"2505.04204","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-email-spam-detection-leveraging","title":"Advancing Email Spam Detection: Leveraging Zero-Shot Learning and Large Language Models","date":"2025-05-05","arxiv_id":"2505.02362","repositories_listed":1,"syntology":null},{"url":"/paper/quickcharnet-an-efficient-url-classification","title":"QuickCharNet: An Efficient URL Classification Framework for Enhanced Search Engine Optimization","date":"2024-10-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/online-detection-and-infographic-explanation","title":"Online detection and infographic explanation of spam reviews with data drift adaptation","date":"2024-06-21","arxiv_id":"2406.15038","repositories_listed":1,"syntology":null},{"url":"/paper/spamdam-towards-privacy-preserving-and","title":"SpamDam: Towards Privacy-Preserving and Adversary-Resistant SMS Spam Detection","date":"2024-04-15","arxiv_id":"2404.09481","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-value-expressive-text-posts-in","title":"Detecting value-expressive text posts in Russian social media","date":"2023-12-14","arxiv_id":"2312.08968","repositories_listed":1,"syntology":null},{"url":"/paper/text-generation-for-dataset-augmentation-in","title":"Text generation for dataset augmentation in security classification tasks","date":"2023-10-22","arxiv_id":"2310.14429","repositories_listed":1,"syntology":null},{"url":"/paper/high-performance-computing-applied-to","title":"High Performance Computing Applied to Logistic Regression: A CPU and GPU Implementation Comparison","date":"2023-08-19","arxiv_id":"2308.10037","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-the-impact-of-localized-training","title":"Addressing the Impact of Localized Training Data in Graph Neural Networks","date":"2023-07-24","arxiv_id":"2307.12689","repositories_listed":1,"syntology":null},{"url":"/paper/alfred-a-system-for-prompted-weak-supervision","title":"Alfred: A System for Prompted Weak Supervision","date":"2023-05-29","arxiv_id":"2305.18623","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/spam-t5-benchmarking-large-language-models","title":"Spam-T5: Benchmarking Large Language Models for Few-Shot Email Spam Detection","date":"2023-04-03","arxiv_id":"2304.01238","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-user-behavior-with-interaction","title":"Modeling User Behavior With Interaction Networks for Spam Detection","date":"2022-07-21","arxiv_id":"2207.10767","repositories_listed":1,"syntology":null},{"url":"/paper/traditional-and-context-specific-spam","title":"Traditional and context-specific spam detection in low resource settings","date":"2022-06-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tlmote-a-topic-based-language-modelling","title":"TLMOTE: A Topic-based Language Modelling Approach for Text Oversampling","date":"2022-05-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/using-bert-encoding-to-tackle-the-mad-lib","title":"Using BERT Encoding to Tackle the Mad-lib Attack in SMS Spam Detection","date":"2021-07-13","arxiv_id":"2107.06400","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-large-scale-real-time-gnn","title":"Accelerating Large Scale Real-Time GNN Inference using Channel Pruning","date":"2021-05-10","arxiv_id":"2105.04528","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/grasp-a-library-for-extracting-and-exploring","title":"GrASP: A Library for Extracting and Exploring Human-Interpretable Textual Patterns","date":"2021-04-08","arxiv_id":"2104.03958","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/adversarial-robustness-with-non-uniform","title":"Adversarial Robustness with Non-uniform Perturbations","date":"2021-02-24","arxiv_id":"2102.12002","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/leveraging-gpt-2-for-classifying-spam-reviews","title":"Leveraging GPT-2 for Classifying Spam Reviews with Limited Labeled Data via Adversarial Training","date":"2020-12-24","arxiv_id":"2012.13400","repositories_listed":1,"syntology":null},{"url":"/paper/fact-or-factitious-contextualized-opinion-1","title":"Fact or Factitious? Contextualized Opinion Spam Detection","date":"2020-10-29","arxiv_id":"2010.15296","repositories_listed":1,"syntology":null},{"url":"/paper/rank-over-class-the-untapped-potential-of","title":"Rank over Class: The Untapped Potential of Ranking in Natural Language Processing","date":"2020-09-10","arxiv_id":"2009.05160","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}