{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/red-teaming/papers/3","list_of":"/task/red-teaming","task":"Red Teaming","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,251],"of":251,"counts":{"archive_papers_tagged":251,"with_a_code_link":110,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":251,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/red-teaming","prev":"/task/red-teaming/papers/2","next":null,"papers":[{"url":null,"slug":"the-multilingual-alignment-prism-aligning","title":"The Multilingual Alignment Prism: Aligning Global and Local Preferences to Reduce Harm","date":"2024-06-26","arxiv_id":"2406.18682","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-in-red","title":"Leveraging Reinforcement Learning in Red Teaming for Advanced Ransomware Attack Simulations","date":"2024-06-25","arxiv_id":"2406.17576","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversaries-can-misuse-combinations-of-safe","title":"Adversaries Can Misuse Combinations of Safe Models","date":"2024-06-20","arxiv_id":"2406.14595","repositories_listed":0,"syntology":null},{"url":"/paper/finding-safety-neurons-in-large-language","slug":"finding-safety-neurons-in-large-language","title":"Finding Safety Neurons in Large Language Models","date":"2024-06-20","arxiv_id":"2406.14144","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finding-safety-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.14144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14144"}},"official":null}},{"url":null,"slug":"cell-your-model-contrastive-explanation","title":"CELL your Model: Contrastive Explanations for Large Language Models","date":"2024-06-17","arxiv_id":"2406.11785","repositories_listed":0,"syntology":null},{"url":null,"slug":"ruby-teaming-improving-quality-diversity","title":"Ruby Teaming: Improving Quality Diversity Search with Memory for Automated Red Teaming","date":"2024-06-17","arxiv_id":"2406.11654","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-sociotechnical-approach-to-red-teaming","title":"STAR: SocioTechnical Approach to Red Teaming Language Models","date":"2024-06-17","arxiv_id":"2406.11757","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreaking-large-language-models-against","title":"Jailbreaking Large Language Models Against Moderation Guardrails via Cipher Characters","date":"2024-05-30","arxiv_id":"2405.20413","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-alignment-for-vision-language-models","title":"Safety Alignment for Vision Language Models","date":"2024-05-22","arxiv_id":"2405.13581","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-refinements-elicit-resilience-toward","title":"Tiny Refinements Elicit Resilience: Toward Efficient Prefix-Model Against LLM Red-Teaming","date":"2024-05-21","arxiv_id":"2405.12604","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mechanism-based-approach-to-mitigating","title":"A Mechanism-Based Approach to Mitigating Harms from Persuasive Generative AI","date":"2024-04-23","arxiv_id":"2404.15058","repositories_listed":0,"syntology":null},{"url":null,"slug":"culturalteaming-ai-assisted-interactive-red","title":"CulturalTeaming: AI-Assisted Interactive Red-Teaming for Challenging LLMs' (Lack of) Multicultural Knowledge","date":"2024-04-10","arxiv_id":"2404.06664","repositories_listed":0,"syntology":null},{"url":null,"slug":"aurora-m-the-first-open-source-multilingual","title":"Aurora-M: Open Source Continual Pre-training for Multilingual Language and Code","date":"2024-03-30","arxiv_id":"2404.00399","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteralign-iterative-constitutional-alignment","title":"IterAlign: Iterative Constitutional Alignment of Large Language Models","date":"2024-03-27","arxiv_id":"2403.18341","repositories_listed":0,"syntology":null},{"url":null,"slug":"hrlaif-improvements-in-helpfulness-and","title":"HRLAIF: Improvements in Helpfulness and Harmlessness in Open-domain Reinforcement Learning From AI Feedback","date":"2024-03-13","arxiv_id":"2403.08309","repositories_listed":0,"syntology":null},{"url":"/paper/red-teaming-models-for-hyperspectral-image","slug":"red-teaming-models-for-hyperspectral-image","title":"Red Teaming Models for Hyperspectral Image Analysis Using Explainable AI","date":"2024-03-12","arxiv_id":"2403.08017","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/red-teaming-models-for-hyperspectral-image#ran","syntology_url":"https://syntology.ai/paper/2403.08017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08017"}},"official":null}},{"url":null,"slug":"a-safe-harbor-for-ai-evaluation-and-red","title":"A Safe Harbor for AI Evaluation and Red Teaming","date":"2024-03-07","arxiv_id":"2403.04893","repositories_listed":0,"syntology":null},{"url":null,"slug":"attackgnn-red-teaming-gnns-in-hardware","title":"AttackGNN: Red-Teaming GNNs in Hardware Security Using Reinforcement Learning","date":"2024-02-21","arxiv_id":"2402.13946","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-bias-representations-in-llama-2","title":"Investigating Bias Representations in Llama 2 Chat via Activation Steering","date":"2024-02-01","arxiv_id":"2402.00402","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-for-generative-ai-silver-bullet","title":"Red-Teaming for Generative AI: Silver Bullet or Security Theater?","date":"2024-01-29","arxiv_id":"2401.15897","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-red-teaming-in-multimodal-and","title":"Towards Red Teaming in Multimodal and Multilingual Translation","date":"2024-01-29","arxiv_id":"2401.16247","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-cloning-of-online-social-networks-for","title":"Digital cloning of online social networks for language-sensitive agent-based modeling of misinformation spread","date":"2024-01-23","arxiv_id":"2401.12509","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-visual-language-models","title":"Red Teaming Visual Language Models","date":"2024-01-23","arxiv_id":"2401.12915","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-red-teaming-framework-for-securing-ai-in","title":"A Red Teaming Framework for Securing AI in Maritime Autonomous Systems","date":"2023-12-08","arxiv_id":"2312.11500","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptprompt-exploiting-llm-driven-code","title":"DeceptPrompt: Exploiting LLM-driven Code Generation via Adversarial Natural Language Instructions","date":"2023-12-07","arxiv_id":"2312.04730","repositories_listed":0,"syntology":null},{"url":null,"slug":"infopattern-unveiling-information-propagation","title":"InfoPattern: Unveiling Information Propagation Patterns in Social Media","date":"2023-11-27","arxiv_id":"2311.15642","repositories_listed":0,"syntology":null},{"url":null,"slug":"jab-joint-adversarial-prompting-and-belief","title":"JAB: Joint Adversarial Prompting and Belief Augmentation","date":"2023-11-16","arxiv_id":"2311.09473","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-exploitability-of-reinforcement","title":"RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models","date":"2023-11-16","arxiv_id":"2311.09641","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreaking-gpt-4v-via-self-adversarial","title":"Jailbreaking GPT-4V via Self-Adversarial Attacks with System Prompts","date":"2023-11-15","arxiv_id":"2311.09127","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-publicly-accountable-frontier-llms","title":"Towards Publicly Accountable Frontier LLMs: Building an External Scrutiny Ecosystem under the ASPIRE Framework","date":"2023-11-15","arxiv_id":"2311.14711","repositories_listed":0,"syntology":null},{"url":"/paper/aart-ai-assisted-red-teaming-with-diverse","slug":"aart-ai-assisted-red-teaming-with-diverse","title":"AART: AI-Assisted Red-Teaming with Diverse Data Generation for New LLM-powered Applications","date":"2023-11-14","arxiv_id":"2311.08592","repositories_listed":0,"syntology":null},{"url":null,"slug":"mart-improving-llm-safety-with-multi-round","title":"MART: Improving LLM Safety with Multi-round Automatic Red-Teaming","date":"2023-11-13","arxiv_id":"2311.07689","repositories_listed":0,"syntology":null},{"url":null,"slug":"summon-a-demon-and-bind-it-a-grounded-theory","title":"Summon a Demon and Bind it: A Grounded Theory of LLM Red Teaming","date":"2023-11-10","arxiv_id":"2311.06237","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-fine-tuning-efficiently-undoes-safety","title":"LoRA Fine-tuning Efficiently Undoes Safety Training in Llama 2-Chat 70B","date":"2023-10-31","arxiv_id":"2310.20624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-red-teaming-gender-bias","title":"Learning from Red Teaming: Gender Bias Provocation and Mitigation in Large Language Models","date":"2023-10-17","arxiv_id":"2310.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-language-models-be-instructed-to-protect","title":"Can Language Models be Instructed to Protect Personal Information?","date":"2023-10-03","arxiv_id":"2310.02224","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-languages-jailbreak-gpt-4","title":"Low-Resource Languages Jailbreak GPT-4","date":"2023-10-03","arxiv_id":"2310.02446","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-generative-ai-nlp-the-bb84","title":"Red Teaming Generative AI/NLP, the BB84 quantum cryptography protocol and the NIST-approved Quantum-Resistant Cryptographic Algorithms","date":"2023-09-17","arxiv_id":"2310.04425","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-promise-and-peril-of-artificial","title":"The Promise and Peril of Artificial Intelligence -- Violet Teaming Offers a Balanced Path Forward","date":"2023-08-28","arxiv_id":"2308.14253","repositories_listed":0,"syntology":null},{"url":null,"slug":"flirt-feedback-loop-in-context-red-teaming","title":"FLIRT: Feedback Loop In-context Red Teaming","date":"2023-08-08","arxiv_id":"2308.04265","repositories_listed":0,"syntology":null},{"url":"/paper/model-card-and-evaluations-for-claude-models","slug":"model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-seeds-beyond-weeds-green-teaming","title":"Seeing Seeds Beyond Weeds: Green Teaming Generative AI for Beneficial Uses","date":"2023-05-30","arxiv_id":"2306.03097","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalisation-within-bounds-a-risk-taxonomy","title":"Personalisation within bounds: A risk taxonomy and policy framework for the alignment of large language models with personalised feedback","date":"2023-03-09","arxiv_id":"2303.05453","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-ai-ethics-of-chatgpt-a-diagnostic","title":"Red teaming ChatGPT via Jailbreaking: Bias, Robustness, Reliability and Toxicity","date":"2023-01-30","arxiv_id":"2301.12867","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-language-models-change-user","title":"Can Large Language Models Change User Preference Adversarially?","date":"2023-01-05","arxiv_id":"2302.10291","repositories_listed":0,"syntology":null},{"url":null,"slug":"red-teaming-the-stable-diffusion-safety","title":"Red-Teaming the Stable Diffusion Safety Filter","date":"2022-10-03","arxiv_id":"2210.04610","repositories_listed":0,"syntology":null},{"url":null,"slug":"cti4ai-threat-intelligence-generation-and","title":"CTI4AI: Threat Intelligence Generation and Sharing after Red Teaming AI Models","date":"2022-08-16","arxiv_id":"2208.07476","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-privilege-escalation-with-deep","title":"Automating Privilege Escalation with Deep Reinforcement Learning","date":"2021-10-04","arxiv_id":"2110.01362","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-disciplinary-review-of-knowledge","title":"A Multi-Disciplinary Review of Knowledge Acquisition Methods: From Human to Autonomous Eliciting Agents","date":"2018-02-27","arxiv_id":"1802.09669","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-red-teaming-in-a-sudoku-solving","title":"Computational Red Teaming in a Sudoku Solving Context: Neural Network Based Skill Representation and Acquisition","date":"2018-02-27","arxiv_id":"1802.09660","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-influence-and-influencing-shaping-a","title":"Shaping Influence and Influencing Shaping: A Computational Red Teaming Trust-based Swarm Intelligence Model","date":"2018-02-26","arxiv_id":"1802.09647","repositories_listed":0,"syntology":null}],"record_sha256":"508c5d6d71a33b51ab435ab7588c5d36d2cc7af31c85ee9c905f8da1db7df4a4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}