{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/safety-alignment/papers/3","list_of":"/task/safety-alignment","task":"Safety Alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,288],"of":288,"counts":{"archive_papers_tagged":288,"with_a_code_link":134,"where_syntology_ran_a_sample":73,"not_listed_spam_title":0,"listed":288,"listed_where_code_ran":73,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/safety-alignment","prev":"/task/safety-alignment/papers/2","next":null,"papers":[{"url":null,"slug":"llama-3-1-sherkala-8b-chat-an-open-large","title":"Llama-3.1-Sherkala-8B-Chat: An Open Large Language Model for Kazakh","date":"2025-03-03","arxiv_id":"2503.01493","repositories_listed":0,"syntology":null},{"url":null,"slug":"fc-attack-jailbreaking-large-vision-language","title":"FC-Attack: Jailbreaking Multimodal Large Language Models via Auto-Generated Flowcharts","date":"2025-02-28","arxiv_id":"2502.21059","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-geometry-of-refusal-in-large-language","title":"The Geometry of Refusal in Large Language Models: Concept Cones and Representational Independence","date":"2025-02-24","arxiv_id":"2502.17420","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-eclipse-manipulating-attention-to","title":"Attention Eclipse: Manipulating Attention to Bypass LLM Safety-Alignment","date":"2025-02-21","arxiv_id":"2502.15334","repositories_listed":0,"syntology":null},{"url":null,"slug":"c3ai-crafting-and-evaluating-constitutions","title":"C3AI: Crafting and Evaluating Constitutions for Constitutional AI","date":"2025-02-21","arxiv_id":"2502.15861","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-safeguarded-ships-run-aground-aligned","title":"Why Safeguarded Ships Run Aground? Aligned Large Language Models' Safety Mechanisms Tend to Be Anchored in The Template Region","date":"2025-02-19","arxiv_id":"2502.13946","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-rectifying-safety","title":"Understanding and Rectifying Safety Perception Distortion in VLMs","date":"2025-02-18","arxiv_id":"2502.13095","repositories_listed":0,"syntology":null},{"url":null,"slug":"delman-dynamic-defense-against-large-language","title":"DELMAN: Dynamic Defense Against Large Language Model Jailbreaking with Model Editing","date":"2025-02-17","arxiv_id":"2502.11647","repositories_listed":0,"syntology":null},{"url":null,"slug":"equilibrate-rlhf-towards-balancing","title":"Equilibrate RLHF: Towards Balancing Helpfulness-Safety Trade-off in Large Language Models","date":"2025-02-17","arxiv_id":"2502.11555","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-guard-safeguarding-vision-language-models","title":"VLM-Guard: Safeguarding Vision-Language Models via Fulfilling Safety Alignment Gap","date":"2025-02-14","arxiv_id":"2502.10486","repositories_listed":0,"syntology":null},{"url":null,"slug":"trustworthy-ai-on-safety-bias-and-privacy-a","title":"Trustworthy AI: Safety, Bias, and Privacy -- A Survey","date":"2025-02-11","arxiv_id":"2502.10450","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-alignment-at-your-discretion","title":"AI Alignment at Your Discretion","date":"2025-02-10","arxiv_id":"2502.10441","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-positive-and-toxic-samples-for-dual","title":"Refining Positive and Toxic Samples for Dual Safety Self-Alignment of LLMs with Minimal Human Interventions","date":"2025-02-08","arxiv_id":"2502.08657","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulnerability-mitigation-for-safety-aligned","title":"Vulnerability Mitigation for Safety-Aligned Language Models via Debiasing","date":"2025-02-04","arxiv_id":"2502.02153","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-activation-as-the-polar-star-for","title":"Internal Activation as the Polar Star for Steering Unsafe LLM Behavior","date":"2025-02-03","arxiv_id":"2502.01042","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dark-deep-side-of-deepseek-fine-tuning","title":"The dark deep side of DeepSeek: Fine-tuning attacks against the safety alignment of CoT-enabled models","date":"2025-02-03","arxiv_id":"2502.01225","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-model-defense-against-jailbreaks","title":"Enhancing Model Defense Against Jailbreaks with Proactive Safety Reasoning","date":"2025-01-31","arxiv_id":"2501.19180","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-ai-clinicians-a-comprehensive","title":"Towards Safe AI Clinicians: A Comprehensive Study on Large Language Model Jailbreaking in Healthcare","date":"2025-01-27","arxiv_id":"2501.18632","repositories_listed":0,"syntology":null},{"url":null,"slug":"tune-in-act-up-exploring-the-impact-of-audio","title":"Jailbreak-AudioBench: In-Depth Evaluation and Analysis of Jailbreak Threats for Large Audio Language Models","date":"2025-01-23","arxiv_id":"2501.13772","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptguard-soft-prompt-guided-unsafe-content","title":"PromptGuard: Soft Prompt-Guided Unsafe Content Moderation for Text-to-Image Models","date":"2025-01-07","arxiv_id":"2501.03544","repositories_listed":0,"syntology":null},{"url":null,"slug":"salora-safety-alignment-preserved-low-rank","title":"SaLoRA: Safety-Alignment Preserved Low-Rank Adaptation","date":"2025-01-03","arxiv_id":"2501.01765","repositories_listed":0,"syntology":null},{"url":null,"slug":"spa-vl-a-comprehensive-safety-preference-1","title":"SPA-VL: A Comprehensive Safety Preference Alignment Dataset for Vision Language Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"no-free-lunch-for-defending-against","title":"No Free Lunch for Defending Against Prefilling Attack by In-Context Learning","date":"2024-12-13","arxiv_id":"2412.12192","repositories_listed":0,"syntology":null},{"url":null,"slug":"safetydpo-scalable-safety-alignment-for-text","title":"SafetyDPO: Scalable Safety Alignment for Text-to-Image Generation","date":"2024-12-13","arxiv_id":"2412.10493","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-editing-based-jailbreak-against-safety","title":"Model-Editing-Based Jailbreak against Safety-aligned Large Language Models","date":"2024-12-11","arxiv_id":"2412.08201","repositories_listed":0,"syntology":null},{"url":null,"slug":"na-vi-or-knave-jailbreaking-language-models","title":"Na'vi or Knave: Jailbreaking Language Models via Metaphorical Avatars","date":"2024-12-10","arxiv_id":"2412.12145","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-alignment-backfires-preventing-the-re","title":"Safety Alignment Backfires: Preventing the Re-emergence of Suppressed Concepts in Fine-tuned Text-to-Image Diffusion Models","date":"2024-11-30","arxiv_id":"2412.00357","repositories_listed":0,"syntology":null},{"url":null,"slug":"peft-as-an-attack-jailbreaking-language","title":"PEFT-as-an-Attack! Jailbreaking Language Models during Federated Parameter-Efficient Fine-Tuning","date":"2024-11-28","arxiv_id":"2411.19335","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-visual-vulnerabilities-via-multi","title":"Exploring Visual Vulnerabilities via Multi-Loss Adversarial Search for Jailbreaking Vision-Language Models","date":"2024-11-27","arxiv_id":"2411.18000","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensuring-safety-and-trust-analyzing-the-risks","title":"Ensuring Safety and Trust: Analyzing the Risks of Large Language Models in Medicine","date":"2024-11-20","arxiv_id":"2411.14487","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-vision-language-model-safety","title":"PSA-VLM: Enhancing Vision-Language Model Safety through Progressive Concept-Bottleneck-Driven Alignment","date":"2024-11-18","arxiv_id":"2411.11543","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-language-game-with-llms-leads-to","title":"Playing Language Game with LLMs Leads to Jailbreaking","date":"2024-11-16","arxiv_id":"2411.12762","repositories_listed":0,"syntology":null},{"url":null,"slug":"unfair-alignment-examining-safety-alignment","title":"Unfair Alignment: Examining Safety Alignment Across Vision Encoder Layers in Vision-Language Models","date":"2024-11-06","arxiv_id":"2411.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-curriculum-learning-for","title":"Code-Switching Curriculum Learning for Multilingual Transfer in LLMs","date":"2024-11-04","arxiv_id":"2411.02460","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaller-large-language-models-can-do-moral","title":"Smaller Large Language Models Can Do Moral Self-Correction","date":"2024-10-30","arxiv_id":"2410.23496","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-the-fragility-of","title":"Towards Understanding the Fragility of Multilingual LLMs against Fine-Tuning Attacks","date":"2024-10-23","arxiv_id":"2410.18210","repositories_listed":0,"syntology":null},{"url":null,"slug":"spin-self-supervised-prompt-injection","title":"SPIN: Self-Supervised Prompt INjection","date":"2024-10-17","arxiv_id":"2410.13236","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-safety-alignment-inference-time","title":"Controllable Safety Alignment: Inference-Time Adaptation to Diverse Safety Requirements","date":"2024-10-11","arxiv_id":"2410.08968","repositories_listed":0,"syntology":null},{"url":null,"slug":"unraveling-and-mitigating-safety-alignment","title":"Unraveling and Mitigating Safety Alignment Degradation of Vision-Language Models","date":"2024-10-11","arxiv_id":"2410.09047","repositories_listed":0,"syntology":null},{"url":null,"slug":"superficial-safety-alignment-hypothesis","title":"Superficial Safety Alignment Hypothesis","date":"2024-10-07","arxiv_id":"2410.10862","repositories_listed":0,"syntology":null},{"url":null,"slug":"toxic-subword-pruning-for-dialogue-response","title":"Toxic Subword Pruning for Dialogue Response Generation on Large Language Models","date":"2024-10-05","arxiv_id":"2410.04155","repositories_listed":0,"syntology":null},{"url":null,"slug":"safeguard-is-a-double-edged-sword-denial-of","title":"LLM Safeguard is a Double-Edged Sword: Exploiting False Positives for Denial-of-Service Attacks","date":"2024-10-03","arxiv_id":"2410.02916","repositories_listed":0,"syntology":null},{"url":null,"slug":"scisafeeval-a-comprehensive-benchmark-for","title":"SciSafeEval: A Comprehensive Benchmark for Safety Alignment of Large Language Models in Scientific Tasks","date":"2024-10-02","arxiv_id":"2410.03769","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-inference-time-category-wise-safety","title":"Towards Inference-time Category-wise Safety Steering for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01174","repositories_listed":0,"syntology":null},{"url":null,"slug":"backtracking-improves-generation-safety","title":"Backtracking Improves Generation Safety","date":"2024-09-22","arxiv_id":"2409.14586","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathseeker-exploring-llm-security","title":"PathSeeker: Exploring LLM Security Vulnerabilities with a Reinforcement Learning-Based Jailbreak Approach","date":"2024-09-21","arxiv_id":"2409.14177","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-against-reverse-preference-attacks","title":"Mitigating Unsafe Feedback with Learning Constraints","date":"2024-09-19","arxiv_id":"2409.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-safety-response-boundary-of-large","title":"Probing the Safety Response Boundary of Large Language Models via Unsafe Decoding Path Generation","date":"2024-08-20","arxiv_id":"2408.10668","repositories_listed":0,"syntology":null},{"url":null,"slug":"sage-rt-synthetic-alignment-data-generation","title":"SAGE-RT: Synthetic Alignment data Generation for Safety Evaluation and Red Teaming","date":"2024-08-14","arxiv_id":"2408.11851","repositories_listed":0,"syntology":null},{"url":null,"slug":"enja-ensemble-jailbreak-on-large-language","title":"EnJa: Ensemble Jailbreak on Large Language Models","date":"2024-08-07","arxiv_id":"2408.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-language-models-automatically","title":"Can Large Language Models Automatically Jailbreak GPT-4V?","date":"2024-07-23","arxiv_id":"2407.16686","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-do-universal-image-jailbreaks-transfer","title":"Failures to Find Transferable Image Jailbreaks Between Vision-Language Models","date":"2024-07-21","arxiv_id":"2407.15211","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-blending-llm-safety-alignment","title":"Multilingual Blending: LLM Safety Alignment Evaluation with Language Mixture","date":"2024-07-10","arxiv_id":"2407.07342","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreak-attacks-and-defenses-against-large","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","date":"2024-07-05","arxiv_id":"2407.04295","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-guard-parameter-efficient-guardrail","title":"LoRA-Guard: Parameter-Efficient Guardrail Adaptation for Content Moderation of Large Language Models","date":"2024-07-03","arxiv_id":"2407.02987","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multilingual-alignment-prism-aligning","title":"The Multilingual Alignment Prism: Aligning Global and Local Preferences to Reduce Harm","date":"2024-06-26","arxiv_id":"2406.18682","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-contrastive-decoding-boosting","title":"Adversarial Contrastive Decoding: Boosting Safety Alignment of Large Language Models via Opposite Prompt Optimization","date":"2024-06-24","arxiv_id":"2406.16743","repositories_listed":0,"syntology":null},{"url":"/paper/finding-safety-neurons-in-large-language","slug":"finding-safety-neurons-in-large-language","title":"Finding Safety Neurons in Large Language Models","date":"2024-06-20","arxiv_id":"2406.14144","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finding-safety-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.14144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14144"}},"official":null}},{"url":null,"slug":"model-merging-and-safety-alignment-one-bad","title":"Model Merging and Safety Alignment: One Bad Model Spoils the Bunch","date":"2024-06-20","arxiv_id":"2406.14563","repositories_listed":0,"syntology":null},{"url":null,"slug":"pku-saferlhf-a-safety-alignment-preference","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","date":"2024-06-20","arxiv_id":"2406.15513","repositories_listed":0,"syntology":null},{"url":null,"slug":"emerging-safety-attack-and-defense-in","title":"Emerging Safety Attack and Defense in Federated Instruction Tuning of Large Language Models","date":"2024-06-15","arxiv_id":"2406.10630","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-user-data-on-mitigating-fine-tuning","title":"Mimicking User Data: On Mitigating Fine-Tuning Risks in Closed Large Language Models","date":"2024-06-12","arxiv_id":"2406.10288","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfdefend-llms-can-defend-themselves-against","title":"SelfDefend: LLMs Can Defend Themselves against Jailbreaking in a Practical Manner","date":"2024-06-08","arxiv_id":"2406.05498","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-intrinsic-self-correction-capability","title":"On the Intrinsic Self-Correction Capability of LLMs: Uncertainty and Latent Concept","date":"2024-06-04","arxiv_id":"2406.02378","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-jailbreak-attack-against-large","title":"Enhancing Jailbreak Attack Against Large Language Models through Silent Tokens","date":"2024-05-31","arxiv_id":"2405.20653","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-safety-alignment-is-textual","title":"Cross-Modal Safety Alignment: Is textual unlearning all you need?","date":"2024-05-27","arxiv_id":"2406.02575","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-two-devils-alike-unveiling-distinct","title":"No Two Devils Alike: Unveiling Distinct Mechanisms of Fine-tuning Attacks","date":"2024-05-25","arxiv_id":"2405.16229","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-safety-aligned-large-language","title":"Robustifying Safety-Aligned Large Language Models through Clean Data Curation","date":"2024-05-24","arxiv_id":"2405.19358","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-alignment-for-vision-language-models","title":"Safety Alignment for Vision Language Models","date":"2024-05-22","arxiv_id":"2405.13581","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-comprehensive-and-efficient-post","title":"Towards Comprehensive Post Safety Alignment of Large Language Models via Safety Patching","date":"2024-05-22","arxiv_id":"2405.13820","repositories_listed":0,"syntology":null},{"url":null,"slug":"wordgame-efficient-effective-llm-jailbreak","title":"WordGame: Efficient & Effective LLM Jailbreak via Simultaneous Obfuscation in Query and Response","date":"2024-05-22","arxiv_id":"2405.14023","repositories_listed":0,"syntology":null},{"url":null,"slug":"canttalkaboutthis-aligning-language-models-to","title":"CantTalkAboutThis: Aligning Language Models to Stay on Topic in Dialogues","date":"2024-04-04","arxiv_id":"2404.03820","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-disguise-avoid-refusal-responses-in","title":"Learn to Disguise: Avoid Refusal Responses in LLM's Defense via a Multi-agent Attacker-Disguiser Game","date":"2024-04-03","arxiv_id":"2404.02532","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpp-based-adversarial-prompt-searching-for","title":"Enhancing Jailbreak Attacks with Diversity Guidance","date":"2024-03-01","arxiv_id":"2403.00292","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-can-defend-themselves-against","title":"LLMs Can Defend Themselves Against Jailbreaking in a Practical Manner: A Vision Paper","date":"2024-02-24","arxiv_id":"2402.15727","repositories_listed":0,"syntology":null},{"url":null,"slug":"break-the-breakout-reinventing-lm-defense","title":"Break the Breakout: Reinventing LM Defense Against Jailbreak Attacks with Self-Refinement","date":"2024-02-23","arxiv_id":"2402.15180","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-brittleness-of-safety-alignment","title":"Assessing the Brittleness of Safety Alignment via Pruning and Low-Rank Modifications","date":"2024-02-07","arxiv_id":"2402.05162","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-overload-jailbreaking-large","title":"Cognitive Overload: Jailbreaking Large Language Models with Overloaded Logical Thinking","date":"2023-11-16","arxiv_id":"2311.09827","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-exploitability-of-reinforcement","title":"RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models","date":"2023-11-16","arxiv_id":"2311.09641","repositories_listed":0,"syntology":null},{"url":null,"slug":"mart-improving-llm-safety-with-multi-round","title":"MART: Improving LLM Safety with Multi-round Automatic Red-Teaming","date":"2023-11-13","arxiv_id":"2311.07689","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-fine-tuning-efficiently-undoes-safety","title":"LoRA Fine-tuning Efficiently Undoes Safety Training in Llama 2-Chat 70B","date":"2023-10-31","arxiv_id":"2310.20624","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-vulnerabilities-in-large-language","title":"Survey of Vulnerabilities in Large Language Models Revealed by Adversarial Attacks","date":"2023-10-16","arxiv_id":"2310.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreak-and-guard-aligned-language-models","title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations","date":"2023-10-10","arxiv_id":"2310.06387","repositories_listed":0,"syntology":null},{"url":null,"slug":"shadow-alignment-the-ease-of-subverting","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","date":"2023-10-04","arxiv_id":"2310.02949","repositories_listed":0,"syntology":null},{"url":null,"slug":"jais-and-jais-chat-arabic-centric-foundation","title":"Jais and Jais-chat: Arabic-Centric Foundation and Instruction-Tuned Open Generative Large Language Models","date":"2023-08-30","arxiv_id":"2308.16149","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-alignment-monitoring","title":"Deceptive Alignment Monitoring","date":"2023-07-20","arxiv_id":"2307.10569","repositories_listed":0,"syntology":null},{"url":"/paper/model-card-and-evaluations-for-claude-models","slug":"model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-risk-assessment-in-markov-decision","title":"Off-Policy Risk Assessment in Markov Decision Processes","date":"2022-09-21","arxiv_id":"2209.10444","repositories_listed":0,"syntology":null}],"record_sha256":"2f89bb2120ed4e94518f310ad3333723f2eca9a4f8c56d9ca173b9864673c34f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}