{"url":"/task/safety-alignment","name":"Safety Alignment","slug":"safety-alignment","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":288,"papers_with_code":134,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/sudo-dataset","name":"SUDO Dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":134,"tagged_in_all":288,"items":[{"url":"/paper/the-hidden-dimensions-of-llm-alignment-a","title":"The Hidden Dimensions of LLM Alignment: A Multi-Dimensional Safety Analysis","date":"2025-02-13","arxiv_id":"2502.09674","repositories_listed":3,"syntology":null},{"url":"/paper/antidote-post-fine-tuning-safety-alignment","title":"Antidote: Post-fine-tuning Safety Alignment for Large Language Models against Harmful Fine-tuning","date":"2024-08-18","arxiv_id":"2408.09600","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/derail-yourself-multi-turn-llm-jailbreak","title":"Derail Yourself: Multi-turn LLM Jailbreak Attack through Self-discovered Clues","date":"2024-10-14","arxiv_id":"2410.10700","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/harmful-fine-tuning-attacks-and-defenses-for","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","date":"2024-09-26","arxiv_id":"2409.18169","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/exploring-safety-generalization-challenges-of","title":"CodeAttack: Revealing Safety Generalization Challenges of Large Language Models via Code Completion","date":"2024-03-12","arxiv_id":"2403.07865","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/figstep-jailbreaking-large-vision-language","title":"FigStep: Jailbreaking Large Vision-Language Models via Typographic Visual Prompts","date":"2023-11-09","arxiv_id":"2311.05608","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/beyond-one-preference-for-all-multi-objective","title":"Beyond One-Preference-Fits-All Alignment: Multi-Objective Direct Preference Optimization","date":"2023-10-05","arxiv_id":"2310.03708","repositories_listed":2,"syntology":null},{"url":"/paper/red-teaming-large-language-models-using-chain","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","date":"2023-08-18","arxiv_id":"2308.09662","repositories_listed":2,"syntology":null},{"url":"/paper/the-devil-behind-the-mask-an-emergent-safety","title":"The Devil behind the mask: An emergent safety vulnerability of Diffusion LLMs","date":"2025-07-15","arxiv_id":"2507.11097","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/safe-pruning-lora-robust-distance-guided","title":"Safe Pruning LoRA: Robust Distance-Guided Pruning for Safety Alignment in Adaptation of LLMs","date":"2025-06-21","arxiv_id":"2506.18931","repositories_listed":1,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":8}},{"url":"/paper/probe-before-you-talk-towards-black-box","title":"Probe before You Talk: Towards Black-box Defense against Backdoor Unalignment for Large Language Models","date":"2025-06-19","arxiv_id":"2506.16447","repositories_listed":1,"syntology":null},{"url":"/paper/probing-the-robustness-of-large-language","title":"Probing the Robustness of Large Language Models Safety to Latent Perturbations","date":"2025-06-19","arxiv_id":"2506.16078","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-safety-fallback-in-editing-based","title":"Mitigating Safety Fallback in Editing-based Backdoor Injection on LLMs","date":"2025-06-16","arxiv_id":"2506.13285","repositories_listed":1,"syntology":null},{"url":"/paper/monitoring-decomposition-attacks-in-llms-with","title":"Monitoring Decomposition Attacks in LLMs with Lightweight Sequential Monitors","date":"2025-06-12","arxiv_id":"2506.10949","repositories_listed":1,"syntology":null},{"url":"/paper/davsp-safety-alignment-for-large-vision","title":"DAVSP: Safety Alignment for Large Vision-Language Models via Deep Aligned Visual Safety Prompt","date":"2025-06-11","arxiv_id":"2506.09353","repositories_listed":1,"syntology":null},{"url":"/paper/rsafe-incentivizing-proactive-reasoning-to","title":"RSafe: Incentivizing proactive reasoning to build robust and adaptive LLM safeguards","date":"2025-06-09","arxiv_id":"2506.07736","repositories_listed":1,"syntology":null},{"url":"/paper/chasing-moving-targets-with-online-self-play","title":"Chasing Moving Targets with Online Self-Play Reinforcement Learning for Safer Language Models","date":"2025-06-09","arxiv_id":"2506.07468","repositories_listed":1,"syntology":{"n":17,"n_ran":6,"n_unverified":11,"n_pointer_only":1}},{"url":"/paper/bitbypass-a-new-direction-in-jailbreaking","title":"BitBypass: A New Direction in Jailbreaking Aligned Large Language Models with Bitstream Camouflage","date":"2025-06-03","arxiv_id":"2506.02479","repositories_listed":1,"syntology":null},{"url":"/paper/diablo-diagonal-blocks-are-sufficient-for","title":"DiaBlo: Diagonal Blocks Are Sufficient For Finetuning","date":"2025-06-03","arxiv_id":"2506.03230","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/trident-enhancing-large-language-model-safety","title":"TRIDENT: Enhancing Large Language Model Safety with Tri-Dimensional Diversified Red-Teaming Data Synthesis","date":"2025-05-30","arxiv_id":"2505.24672","repositories_listed":1,"syntology":null},{"url":"/paper/agentalign-navigating-safety-alignment-in-the","title":"AgentAlign: Navigating Safety Alignment in the Shift from Informative to Agentic Large Language Models","date":"2025-05-29","arxiv_id":"2505.23020","repositories_listed":1,"syntology":null},{"url":"/paper/overt-a-benchmark-for-over-refusal-evaluation","title":"OVERT: A Benchmark for Over-Refusal Evaluation on Text-to-Image Models","date":"2025-05-27","arxiv_id":"2505.21347","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/lifelong-safety-alignment-for-language-models","title":"Lifelong Safety Alignment for Language Models","date":"2025-05-26","arxiv_id":"2505.20259","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-safe-answers-a-benchmark-for","title":"Beyond Safe Answers: A Benchmark for Evaluating True Risk Awareness in Large Reasoning Models","date":"2025-05-26","arxiv_id":"2505.19690","repositories_listed":1,"syntology":null},{"url":"/paper/vscbench-bridging-the-gap-in-vision-language","title":"VSCBench: Bridging the Gap in Vision-Language Model Safety Calibration","date":"2025-05-26","arxiv_id":"2505.20362","repositories_listed":1,"syntology":null},{"url":"/paper/one-model-transfer-to-all-on-robust-jailbreak","title":"One Model Transfer to All: On Robust Jailbreak Prompts Generation against LLMs","date":"2025-05-23","arxiv_id":"2505.17598","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-fine-tuning-risks-in-llms-via","title":"Mitigating Fine-tuning Risks in LLMs via Safety-Aware Probing Optimization","date":"2025-05-22","arxiv_id":"2505.16737","repositories_listed":1,"syntology":null},{"url":"/paper/duffin-a-dual-level-fingerprinting-framework","title":"DuFFin: A Dual-Level Fingerprinting Framework for LLMs IP Protection","date":"2025-05-22","arxiv_id":"2505.16530","repositories_listed":1,"syntology":null},{"url":"/paper/mpo-multilingual-safety-alignment-via-reward","title":"MPO: Multilingual Safety Alignment via Reward Gap Optimization","date":"2025-05-22","arxiv_id":"2505.16869","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/mtsa-multi-turn-safety-alignment-for-llms","title":"MTSA: Multi-turn Safety Alignment for LLMs through Multi-round Red-teaming","date":"2025-05-22","arxiv_id":"2505.17147","repositories_listed":1,"syntology":null}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}