{"url":"/task/red-teaming","name":"Red Teaming","slug":"red-teaming","description_markdown":null,"categories":[{"name":"Adversarial","url":"/area/adversarial"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":251,"papers_with_code":110,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/red-teaming-on-sudo-dataset","slug":"red-teaming-on-sudo-dataset","dataset":"SUDO Dataset","dataset_url":"/dataset/sudo-dataset","rows_in_archive":1,"metrics":["Attack Success Rate"],"first_row_in_archive_order":{"model":"SUDO","paper_title":"sudo rm -rf agentic_security","paper_url":"/paper/sudo-rm-rf-agentic-security","paper_date":"2025-03-26","arxiv_id":"2503.20279","code_links":[{"title":"AIM-Intelligence/SUDO","url":"https://github.com/AIM-Intelligence/SUDO"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}}],"datasets":[{"url":"/dataset/sudo-dataset","name":"SUDO Dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":110,"tagged_in_all":251,"items":[{"url":"/paper/wildteaming-at-scale-from-in-the-wild","title":"WildTeaming at Scale: From In-the-Wild Jailbreaks to (Adversarially) Safer Language Models","date":"2024-06-26","arxiv_id":"2406.18510","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/harmbench-a-standardized-evaluation-framework","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","date":"2024-02-06","arxiv_id":"2402.04249","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/gptfuzzer-red-teaming-large-language-models","title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","date":"2023-09-19","arxiv_id":"2309.10253","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/explore-establish-exploit-red-teaming","title":"Explore, Establish, Exploit: Red Teaming Language Models from Scratch","date":"2023-06-15","arxiv_id":"2306.09442","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/autodan-turbo-a-lifelong-agent-for-strategy","title":"AutoDAN-Turbo: A Lifelong Agent for Strategy Self-Exploration to Jailbreak LLMs","date":"2024-10-03","arxiv_id":"2410.05295","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/2408-00761","title":"Tamper-Resistant Safeguards for Open-Weight LLMs","date":"2024-08-01","arxiv_id":"2408.00761","repositories_listed":2,"syntology":{"n":16,"n_ran":9,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/targeted-latent-adversarial-training-improves","title":"Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in LLMs","date":"2024-07-22","arxiv_id":"2407.15549","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/garak-a-framework-for-security-probing-large","title":"garak: A Framework for Security Probing Large Language Models","date":"2024-06-16","arxiv_id":"2406.11036","repositories_listed":2,"syntology":null},{"url":"/paper/alert-a-comprehensive-benchmark-for-assessing","title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming","date":"2024-04-06","arxiv_id":"2404.08676","repositories_listed":2,"syntology":null},{"url":"/paper/defending-against-unforeseen-failure-modes","title":"Defending Against Unforeseen Failure Modes with Latent Adversarial Training","date":"2024-03-08","arxiv_id":"2403.05030","repositories_listed":2,"syntology":null},{"url":"/paper/aligners-decoupling-llms-and-alignment","title":"Aligners: Decoupling LLMs and Alignment","date":"2024-03-07","arxiv_id":"2403.04224","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ring-a-bell-how-reliable-are-concept-removal","title":"Ring-A-Bell! How Reliable are Concept Removal Methods for Diffusion Models?","date":"2023-10-16","arxiv_id":"2310.10012","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/large-language-model-unlearning","title":"Large Language Model Unlearning","date":"2023-10-14","arxiv_id":"2310.10683","repositories_listed":2,"syntology":null},{"url":"/paper/catastrophic-jailbreak-of-open-source-llms","title":"Catastrophic Jailbreak of Open-source LLMs via Exploiting Generation","date":"2023-10-10","arxiv_id":"2310.06987","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}},{"url":"/paper/red-teaming-large-language-models-using-chain","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","date":"2023-08-18","arxiv_id":"2308.09662","repositories_listed":2,"syntology":null},{"url":"/paper/red-teaming-language-model-detectors-with","title":"Red Teaming Language Model Detectors with Language Models","date":"2023-05-31","arxiv_id":"2305.19713","repositories_listed":2,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/red-teaming-language-models-to-reduce-harms","title":"Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","date":"2022-08-23","arxiv_id":"2209.07858","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/rabakbench-scaling-human-annotations-to","title":"RabakBench: Scaling Human Annotations to Construct Localized Multilingual Safety Benchmarks for Low-Resource Languages","date":"2025-07-08","arxiv_id":"2507.05980","repositories_listed":1,"syntology":null},{"url":"/paper/we-should-identify-and-mitigate-third-party","title":"We Should Identify and Mitigate Third-Party Safety Risks in MCP-Powered Agent Systems","date":"2025-06-16","arxiv_id":"2506.13666","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/redrft-a-light-weight-benchmark-for","title":"RedRFT: A Light-Weight Benchmark for Reinforcement Fine-Tuning-Based Red Teaming","date":"2025-06-04","arxiv_id":"2506.04302","repositories_listed":1,"syntology":null},{"url":"/paper/reddebate-safer-responses-through-multi-agent","title":"RedDebate: Safer Responses through Multi-Agent Red Teaming Debates","date":"2025-06-04","arxiv_id":"2506.11083","repositories_listed":1,"syntology":null},{"url":"/paper/bitbypass-a-new-direction-in-jailbreaking","title":"BitBypass: A New Direction in Jailbreaking Aligned Large Language Models with Bitstream Camouflage","date":"2025-06-03","arxiv_id":"2506.02479","repositories_listed":1,"syntology":null},{"url":"/paper/trident-enhancing-large-language-model-safety","title":"TRIDENT: Enhancing Large Language Model Safety with Tri-Dimensional Diversified Red-Teaming Data Synthesis","date":"2025-05-30","arxiv_id":"2505.24672","repositories_listed":1,"syntology":null},{"url":"/paper/redteamcua-realistic-adversarial-testing-of","title":"RedTeamCUA: Realistic Adversarial Testing of Computer-Use Agents in Hybrid Web-OS Environments","date":"2025-05-28","arxiv_id":"2505.21936","repositories_listed":1,"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/capability-based-scaling-laws-for-llm-red","title":"Capability-Based Scaling Laws for LLM Red-Teaming","date":"2025-05-26","arxiv_id":"2505.20162","repositories_listed":1,"syntology":{"n":23,"n_ran":0,"n_unverified":23,"n_pointer_only":0}},{"url":"/paper/mtsa-multi-turn-safety-alignment-for-llms","title":"MTSA: Multi-turn Safety Alignment for LLMs through Multi-round Red-teaming","date":"2025-05-22","arxiv_id":"2505.17147","repositories_listed":1,"syntology":null},{"url":"/paper/soft-prompts-for-evaluation-measuring","title":"Soft Prompts for Evaluation: Measuring Conditional Distance of Capabilities","date":"2025-05-20","arxiv_id":"2505.14943","repositories_listed":1,"syntology":null},{"url":"/paper/benign-samples-matter-fine-tuning-on-outlier","title":"Benign Samples Matter! Fine-tuning On Outlier Benign Samples Severely Breaks Safety","date":"2025-05-11","arxiv_id":"2505.06843","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/oet-optimization-based-prompt-injection","title":"OET: Optimization-based prompt injection Evaluation Toolkit","date":"2025-05-01","arxiv_id":"2505.00843","repositories_listed":1,"syntology":null}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}