{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/red-teaming/papers/2","list_of":"/task/red-teaming","task":"Red Teaming","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":251,"counts":{"archive_papers_tagged":251,"with_a_code_link":110,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":251,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/red-teaming","prev":"/task/red-teaming","next":"/task/red-teaming/papers/3","papers":[{"url":"/paper/attack-prompt-generation-for-red-teaming-and","slug":"attack-prompt-generation-for-red-teaming-and","title":"Attack Prompt Generation for Red Teaming and Defending Large Language Models","date":"2023-10-19","arxiv_id":"2310.12505","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attack-prompt-generation-for-red-teaming-and#ran","syntology_url":"https://syntology.ai/paper/2310.12505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12505"}},"official":{"repos":["aatrox103/sap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assert-automated-safety-scenario-red-teaming","slug":"assert-automated-safety-scenario-red-teaming","title":"ASSERT: Automated Safety Scenario Red Teaming for Evaluating the Robustness of Large Language Models","date":"2023-10-14","arxiv_id":"2310.09624","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-aligned-language-models","slug":"fine-tuning-aligned-language-models","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","date":"2023-10-05","arxiv_id":"2310.03693","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fine-tuning-aligned-language-models#ran","syntology_url":"https://syntology.ai/paper/2310.03693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03693"}},"official":{"repos":["llm-tuning-safety/llms-finetuning-safety"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/no-offense-taken-eliciting-offensiveness-from","slug":"no-offense-taken-eliciting-offensiveness-from","title":"No Offense Taken: Eliciting Offensiveness from Language Models","date":"2023-10-02","arxiv_id":"2310.00892","repositories_listed":1,"syntology":null},{"url":"/paper/prompting4debugging-red-teaming-text-to-image","slug":"prompting4debugging-red-teaming-text-to-image","title":"Prompting4Debugging: Red-Teaming Text-to-Image Diffusion Models by Finding Problematic Prompts","date":"2023-09-12","arxiv_id":"2309.06135","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompting4debugging-red-teaming-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2309.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06135"}},"official":{"repos":["joycenerd/p4d"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat","slug":"gpt-4-is-too-smart-to-be-safe-stealthy-chat","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","date":"2023-08-12","arxiv_id":"2308.06463","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat#ran","syntology_url":"https://syntology.ai/paper/2308.06463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06463"}},"official":{"repos":["robustnlp/cipherchat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xstest-a-test-suite-for-identifying","slug":"xstest-a-test-suite-for-identifying","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","date":"2023-08-02","arxiv_id":"2308.01263","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/xstest-a-test-suite-for-identifying#ran","syntology_url":"https://syntology.ai/paper/2308.01263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01263"}},"official":{"repos":["paul-rottger/exaggerated-safety"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jailbroken-how-does-llm-safety-training-fail","slug":"jailbroken-how-does-llm-safety-training-fail","title":"Jailbroken: How Does LLM Safety Training Fail?","date":"2023-07-05","arxiv_id":"2307.02483","repositories_listed":1,"syntology":null},{"url":"/paper/query-efficient-black-box-red-teaming-via","slug":"query-efficient-black-box-red-teaming-via","title":"Query-Efficient Black-Box Red Teaming via Bayesian Optimization","date":"2023-05-27","arxiv_id":"2305.17444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-efficient-black-box-red-teaming-via#ran","syntology_url":"https://syntology.ai/paper/2305.17444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17444"}},"official":{"repos":["snu-mllab/bayesian-red-teaming"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/red-teaming-language-models-with-language","slug":"red-teaming-language-models-with-language","title":"Red Teaming Language Models with Language Models","date":"2022-02-07","arxiv_id":"2202.03286","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/red-teaming-language-models-with-language#ran","syntology_url":"https://syntology.ai/paper/2202.03286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03286"}},"official":null}},{"url":null,"slug":"stack-adversarial-attacks-on-llm-safeguard","title":"STACK: Adversarial Attacks on LLM Safeguard Pipelines","date":"2025-06-30","arxiv_id":"2506.24068","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10047","title":"GenBreak: Red Teaming Text-to-Image Generators Using Large Language Models","date":"2025-06-11","arxiv_id":"2506.10047","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-red-teaming-of-policy-adherent","title":"Effective Red-Teaming of Policy-Adherent Agents","date":"2025-06-11","arxiv_id":"2506.09600","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-diversity-red-teaming-automated","title":"Quality-Diversity Red-Teaming: Automated Generation of High-Quality and Diverse Attackers for Large Language Models","date":"2025-06-08","arxiv_id":"2506.07121","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-ai-policy-a-taxonomy-of-avoision","title":"Red Teaming AI Policy: A Taxonomy of Avoision and the EU AI Act","date":"2025-06-02","arxiv_id":"2506.01931","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-red-teaming-roadmap-towards-system-level","title":"A Red Teaming Roadmap Towards System-Level Safety","date":"2025-05-30","arxiv_id":"2506.05376","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reward-driven-automated-webshell-malicious","title":"A Reward-driven Automated Webshell Malicious-code Generator for Red-teaming","date":"2025-05-30","arxiv_id":"2505.24252","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-secure-mlops-surveying-attacks","title":"Towards Secure MLOps: Surveying Attacks, Mitigation Strategies, and Research Challenges","date":"2025-05-30","arxiv_id":"2506.02032","repositories_listed":0,"syntology":null},{"url":null,"slug":"cot-red-handed-stress-testing-chain-of","title":"CoT Red-Handed: Stress Testing Chain-of-Thought Monitoring","date":"2025-05-29","arxiv_id":"2505.23575","repositories_listed":0,"syntology":null},{"url":null,"slug":"safecomm-what-about-safety-alignment-in-fine","title":"SafeCOMM: What about Safety Alignment in Fine-Tuned Telecom Large Language Models?","date":"2025-05-29","arxiv_id":"2506.00062","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-text-to-image-systems-by-rule","title":"Red-Teaming Text-to-Image Systems by Rule-based Preference Modeling","date":"2025-05-27","arxiv_id":"2505.21074","repositories_listed":0,"syntology":null},{"url":null,"slug":"ghostprompt-jailbreaking-text-to-image","title":"GhostPrompt: Jailbreaking Text-to-image Generative Models based on Dynamic Optimization","date":"2025-05-25","arxiv_id":"2505.18979","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-vulnerability-of-the-content","title":"Exploring the Vulnerability of the Content Moderation Guardrail in Large Language Models via Intent Manipulation","date":"2025-05-24","arxiv_id":"2505.18556","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-medical-ai-misalignment-a-preliminary","title":"Towards medical AI misalignment: a preliminary study","date":"2025-05-22","arxiv_id":"2505.18212","repositories_listed":0,"syntology":null},{"url":null,"slug":"rrtl-red-teaming-reasoning-large-language","title":"RRTL: Red Teaming Reasoning Large Language Models in Tool Learning","date":"2025-05-21","arxiv_id":"2505.17106","repositories_listed":0,"syntology":null},{"url":null,"slug":"eva-red-teaming-gui-agents-via-evolving","title":"EVA: Red-Teaming GUI Agents via Evolving Indirect Prompt Injection","date":"2025-05-20","arxiv_id":"2505.14289","repositories_listed":0,"syntology":null},{"url":null,"slug":"haet-bhasha-aur-diskrimineshun-phonetic","title":"\"Haet Bhasha aur Diskrimineshun\": Phonetic Perturbations in Code-Mixed Hinglish to Red-Team LLMs","date":"2025-05-20","arxiv_id":"2505.14226","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-ghost-hand-unveiling-backdoor","title":"Hidden Ghost Hand: Unveiling Backdoor Vulnerabilities in MLLM-Powered Mobile GUI Agents","date":"2025-05-20","arxiv_id":"2505.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"cure-concept-unlearning-via-orthogonal","title":"CURE: Concept Unlearning via Orthogonal Representation Editing in Diffusion Models","date":"2025-05-19","arxiv_id":"2505.12677","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10838","title":"LARGO: Latent Adversarial Reflection through Gradient Optimization for Jailbreaking LLMs","date":"2025-05-16","arxiv_id":"2505.10838","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentxploit-end-to-end-redteaming-of-black","title":"AgentVigil: Generic Black-Box Red-teaming for Indirect Prompt Injection against LLM Agents","date":"2025-05-09","arxiv_id":"2505.05849","repositories_listed":0,"syntology":null},{"url":null,"slug":"offensive-security-for-ai-systems-concepts","title":"Offensive Security for AI Systems: Concepts, Practices, and Applications","date":"2025-05-09","arxiv_id":"2505.06380","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-by-measurement-a-systematic-literature","title":"Safety by Measurement: A Systematic Literature Review of AI Safety Evaluation Methods","date":"2025-05-08","arxiv_id":"2505.05541","repositories_listed":0,"syntology":null},{"url":null,"slug":"dmrl-data-and-model-aware-reward-learning-for","title":"DMRL: Data- and Model-aware Reward Learning for Data Extraction","date":"2025-05-07","arxiv_id":"2505.06284","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-the-mind-of-the-machine-a","title":"Red Teaming the Mind of the Machine: A Systematic Evaluation of Prompt Injection and Jailbreak Vulnerabilities in LLMs","date":"2025-05-07","arxiv_id":"2505.04806","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-large-language-models-for","title":"Red Teaming Large Language Models for Healthcare","date":"2025-05-01","arxiv_id":"2505.00467","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-testing-ai-tests-us-safeguarding-mental","title":"When Testing AI Tests Us: Safeguarding Mental Health on the Digital Frontlines","date":"2025-04-29","arxiv_id":"2504.20910","repositories_listed":0,"syntology":null},{"url":null,"slug":"rag-llms-are-not-safer-a-safety-analysis-of","title":"RAG LLMs are Not Safer: A Safety Analysis of Retrieval-Augmented Generation for Large Language Models","date":"2025-04-25","arxiv_id":"2504.18041","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-mitigating-risks-of","title":"Understanding and Mitigating Risks of Generative AI in Financial Services","date":"2025-04-25","arxiv_id":"2504.20086","repositories_listed":0,"syntology":null},{"url":null,"slug":"elab-extensive-llm-alignment-benchmark-in","title":"ELAB: Extensive LLM Alignment Benchmark in Persian Language","date":"2025-04-17","arxiv_id":"2504.12553","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-teaming-multi-turn-jailbreaks-and-defenses","title":"X-Teaming: Multi-Turn Jailbreaks and Defenses with Adaptive Multi-Agents","date":"2025-04-15","arxiv_id":"2504.13203","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-lingual-multi-turn-automated-red","title":"Multi-lingual Multi-turn Automated Red Teaming for LLMs","date":"2025-04-04","arxiv_id":"2504.03174","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategize-globally-adapt-locally-a-multi","title":"Strategize Globally, Adapt Locally: A Multi-Turn Red Teaming Agent with Dual-Level Learning","date":"2025-04-02","arxiv_id":"2504.01278","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-with-artificial-intelligence","title":"Red Teaming with Artificial Intelligence-Driven Cyberattacks: A Scoping Review","date":"2025-03-25","arxiv_id":"2503.19626","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoredteamer-autonomous-red-teaming-with","title":"AutoRedTeamer: Autonomous Red Teaming with Lifelong Attack Integration","date":"2025-03-20","arxiv_id":"2503.15754","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmdt-decoding-the-trustworthiness-and-safety","title":"MMDT: Decoding the Trustworthiness and Safety of Multimodal Foundation Models","date":"2025-03-19","arxiv_id":"2503.14827","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-evaluating-emerging","title":"A Framework for Evaluating Emerging Cyberattack Capabilities of AI","date":"2025-03-14","arxiv_id":"2503.11917","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-every-step-effective-jailbreaking","title":"Making Every Step Effective: Jailbreaking Large Vision-Language Models Through Hierarchical KV Equalization","date":"2025-03-14","arxiv_id":"2503.11750","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-contemporary-ai-models-insights","title":"Red Teaming Contemporary AI Models: Insights from Spanish and Basque Perspectives","date":"2025-03-13","arxiv_id":"2503.10192","repositories_listed":0,"syntology":null},{"url":null,"slug":"jbfuzz-jailbreaking-llms-efficiently-and","title":"JBFuzz: Jailbreaking LLMs Efficiently and Effectively Using Fuzzing","date":"2025-03-12","arxiv_id":"2503.08990","repositories_listed":0,"syntology":null},{"url":null,"slug":"mad-max-modular-and-diverse-malicious-attack","title":"MAD-MAX: Modular And Diverse Malicious Attack MiXtures for Automated LLM Red Teaming","date":"2025-03-08","arxiv_id":"2503.06253","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-diffuser-for-red-teaming-large","title":"Reinforced Diffuser for Red Teaming Large Vision-Language Models","date":"2025-03-08","arxiv_id":"2503.06223","repositories_listed":0,"syntology":null},{"url":null,"slug":"know-thy-judge-on-the-robustness-meta","title":"Know Thy Judge: On the Robustness Meta-Evaluation of LLM Safety Judges","date":"2025-03-06","arxiv_id":"2503.04474","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-safety-evaluations-lack-robustness","title":"LLM-Safety Evaluations Lack Robustness","date":"2025-03-04","arxiv_id":"2503.02574","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-safe-genai-applications-an-end-to","title":"Building Safe GenAI Applications: An End-to-End Overview of Red Teaming for Large Language Models","date":"2025-03-03","arxiv_id":"2503.01742","repositories_listed":0,"syntology":null},{"url":null,"slug":"be-a-multitude-to-itself-a-prompt-evolution","title":"Be a Multitude to Itself: A Prompt Evolution Framework for Red Teaming","date":"2025-02-22","arxiv_id":"2502.16109","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-proxies-for-llm-robustness-evaluation","title":"Fast Proxies for LLM Robustness Evaluation","date":"2025-02-14","arxiv_id":"2502.10487","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-frontier-ai-risk-management-framework","title":"A Frontier AI Risk Management Framework: Bridging the Gap Between Current AI Practices and Established Risk Management","date":"2025-02-10","arxiv_id":"2502.06656","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-red-teaming-breaking-policies","title":"Predictive Red Teaming: Breaking Policies Without Breaking Robots","date":"2025-02-10","arxiv_id":"2502.06575","repositories_listed":0,"syntology":null},{"url":null,"slug":"kda-a-knowledge-distilled-attacker-for","title":"KDA: A Knowledge-Distilled Attacker for Generating Diverse Prompts to Jailbreak LLMs","date":"2025-02-05","arxiv_id":"2502.05223","repositories_listed":0,"syntology":null},{"url":null,"slug":"constitutional-classifiers-defending-against","title":"Constitutional Classifiers: Defending against Universal Jailbreaks across Thousands of Hours of Red Teaming","date":"2025-01-31","arxiv_id":"2501.18837","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-devil-s-advocate-unmasking-toxicity","title":"Playing Devil's Advocate: Unmasking Toxicity and Vulnerabilities in Large Vision-Language Models","date":"2025-01-14","arxiv_id":"2501.09039","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-diffusion-red-teaming-of-large-language","title":"Text-Diffusion Red-Teaming of Large Language Models: Unveiling Harmful Behaviors with Proximity Constraints","date":"2025-01-14","arxiv_id":"2501.08246","repositories_listed":0,"syntology":null},{"url":null,"slug":"lessons-from-red-teaming-100-generative-ai","title":"Lessons From Red Teaming 100 Generative AI Products","date":"2025-01-13","arxiv_id":"2501.07238","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreaking-multimodal-large-language-models","title":"Jailbreaking Multimodal Large Language Models via Shuffle Inconsistency","date":"2025-01-09","arxiv_id":"2501.04931","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-rt-automatic-jailbreak-strategy","title":"Auto-RT: Automatic Jailbreak Strategy Exploration for Red-Teaming Large Language Models","date":"2025-01-03","arxiv_id":"2501.01830","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-and-effective-red-teaming-with-auto","title":"Diverse and Effective Red Teaming with Auto-generated Rewards and Multi-step Reinforcement Learning","date":"2024-12-24","arxiv_id":"2412.18693","repositories_listed":0,"syntology":null},{"url":null,"slug":"openai-o1-system-card","title":"OpenAI o1 System Card","date":"2024-12-21","arxiv_id":"2412.16720","repositories_listed":0,"syntology":null},{"url":null,"slug":"poex-policy-executable-embodied-ai-jailbreak","title":"POEX: Understanding and Mitigating Policy Executable Jailbreak Attacks against Embodied AI","date":"2024-12-21","arxiv_id":"2412.16633","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-red-teaming-is-a-sociotechnical-system-now","title":"AI red-teaming is a sociotechnical challenge: on values, labor, and harms","date":"2024-12-12","arxiv_id":"2412.09751","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-red-teaming-for-auditing-robotic","title":"Embodied Red Teaming for Auditing Robotic Foundation Models","date":"2024-11-27","arxiv_id":"2411.18676","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-experience-replay-facilitates","title":"In-Context Experience Replay Facilitates Safety Red-Teaming of Text-to-Image Diffusion Models","date":"2024-11-25","arxiv_id":"2411.16769","repositories_listed":0,"syntology":null},{"url":null,"slug":"llmstinger-jailbreaking-llms-using-rl-fine","title":"LLMStinger: Jailbreaking LLMs using RL fine-tuned LLMs","date":"2024-11-13","arxiv_id":"2411.08862","repositories_listed":0,"syntology":null},{"url":null,"slug":"desert-camels-and-oil-sheikhs-arab-centric","title":"Desert Camels and Oil Sheikhs: Arab-Centric Red Teaming of Frontier LLMs","date":"2024-10-31","arxiv_id":"2410.24049","repositories_listed":0,"syntology":null},{"url":null,"slug":"advweb-controllable-black-box-attacks-on-vlm","title":"AdvAgent: Controllable Blackbox Red-teaming on Web Agents","date":"2024-10-22","arxiv_id":"2410.17401","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-assisted-red-teaming-of-diffusion-models","title":"LLM-Assisted Red Teaming of Diffusion Models through \"Failures Are Fated, But Can Be Faded\"","date":"2024-10-22","arxiv_id":"2410.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"insights-and-current-gaps-in-open-source-llm","title":"Insights and Current Gaps in Open-Source LLM Vulnerability Scanners: A Comparative Analysis","date":"2024-10-21","arxiv_id":"2410.16527","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-formal-framework-for-assessing-and","title":"A Formal Framework for Assessing and Mitigating Emergent Security Risks in Generative AI Models: Bridging Theory and Dynamic Risk Mitigation","date":"2024-10-15","arxiv_id":"2410.13897","repositories_listed":0,"syntology":null},{"url":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset","slug":"vlfeedback-a-large-scale-ai-feedback-dataset","title":"VLFeedback: A Large-Scale AI Feedback Dataset for Large Vision-Language Models Alignment","date":"2024-10-12","arxiv_id":"2410.09421","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.09421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09421"}},"official":null}},{"url":null,"slug":"recent-advancements-in-llm-red-teaming","title":"Recent advancements in LLM Red-Teaming: Techniques, Defenses, and Ethical Considerations","date":"2024-10-09","arxiv_id":"2410.09097","repositories_listed":0,"syntology":null},{"url":null,"slug":"steerdiff-steering-towards-safe-text-to-image","title":"SteerDiff: Steering towards Safe Text-to-Image Diffusion Models","date":"2024-10-03","arxiv_id":"2410.02710","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-red-teaming-with-goat-the","title":"Automated Red Teaming with GOAT: the Generative Offensive Agent Tester","date":"2024-10-02","arxiv_id":"2410.01606","repositories_listed":0,"syntology":null},{"url":null,"slug":"attack-atlas-a-practitioner-s-perspective-on","title":"Attack Atlas: A Practitioner's Perspective on Challenges and Pitfalls in Red Teaming GenAI","date":"2024-09-23","arxiv_id":"2409.15398","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreaking-large-language-models-with","title":"Jailbreaking Large Language Models with Symbolic Mathematics","date":"2024-09-17","arxiv_id":"2409.11445","repositories_listed":0,"syntology":null},{"url":null,"slug":"games-for-ai-control-models-of-safety","title":"Games for AI Control: Models of Safety Evaluations of AI Deployment Protocols","date":"2024-09-12","arxiv_id":"2409.07985","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-straightforward-conversational-red","title":"Exploring Straightforward Conversational Red-Teaming","date":"2024-09-07","arxiv_id":"2409.04822","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-complexity-for-assessing-risk","title":"Conversational Complexity for Assessing Risk in Large Language Models","date":"2024-09-02","arxiv_id":"2409.01247","repositories_listed":0,"syntology":null},{"url":null,"slug":"testing-and-evaluation-of-large-language","title":"Testing and Evaluation of Large Language Models: Correctness, Non-Toxicity, and Fairness","date":"2024-08-31","arxiv_id":"2409.00551","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-ai-flaws-target-driven-attacks-on","title":"Atoxia: Red-teaming Large Language Models with Target Toxic Answers","date":"2024-08-27","arxiv_id":"2408.14853","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffzoo-a-purely-query-based-black-box-attack","title":"DiffZOO: A Purely Query-Based Black-Box Attack for Red-teaming Text-to-Image Generative Model via Zeroth Order Optimization","date":"2024-08-18","arxiv_id":"2408.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"sage-rt-synthetic-alignment-data-generation","title":"SAGE-RT: Synthetic Alignment data Generation for Safety Evaluation and Red Teaming","date":"2024-08-14","arxiv_id":"2408.11851","repositories_listed":0,"syntology":null},{"url":null,"slug":"h4rm3l-a-dynamic-benchmark-of-composable","title":"h4rm3l: A language for Composable Jailbreak Attack Synthesis","date":"2024-08-09","arxiv_id":"2408.04811","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-language-models-automatically","title":"Can Large Language Models Automatically Jailbreak GPT-4V?","date":"2024-07-23","arxiv_id":"2407.16686","repositories_listed":0,"syntology":null},{"url":null,"slug":"redagent-red-teaming-large-language-models","title":"RedAgent: Red Teaming Large Language Models with Context-aware Autonomous Language Agent","date":"2024-07-23","arxiv_id":"2407.16667","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-global-north-stereotype-a-global","title":"Breaking the Global North Stereotype: A Global South-centric Benchmark Dataset for Auditing and Mitigating Biases in Facial Recognition Systems","date":"2024-07-22","arxiv_id":"2407.15810","repositories_listed":0,"syntology":null},{"url":null,"slug":"arondight-red-teaming-large-vision-language","title":"Arondight: Red Teaming Large Vision Language Models with Auto-generated Multi-modal Jailbreak Prompts","date":"2024-07-21","arxiv_id":"2407.15050","repositories_listed":0,"syntology":null},{"url":null,"slug":"phi-3-safety-post-training-aligning-language","title":"Phi-3 Safety Post-Training: Aligning Language Models with a \"Break-Fix\" Cycle","date":"2024-07-18","arxiv_id":"2407.13833","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-unlearning-optimization-for-robust-and","title":"Direct Unlearning Optimization for Robust and Safe Text-to-Image Models","date":"2024-07-17","arxiv_id":"2407.21035","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-human-factor-in-ai-red-teaming","title":"The Human Factor in AI Red Teaming: Perspectives from Social and Collaborative Computing","date":"2024-07-10","arxiv_id":"2407.07786","repositories_listed":0,"syntology":null},{"url":null,"slug":"purple-teaming-llms-with-adversarial-defender","title":"Purple-teaming LLMs with Adversarial Defender Training","date":"2024-07-01","arxiv_id":"2407.01850","repositories_listed":0,"syntology":null}],"record_sha256":"9b294a836f0827c49936ba4df84be346e9e323bc6212da4ab42b9d526271ef7a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}