{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/safety-alignment/papers/2","list_of":"/task/safety-alignment","task":"Safety Alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":288,"counts":{"archive_papers_tagged":288,"with_a_code_link":134,"where_syntology_ran_a_sample":73,"not_listed_spam_title":0,"listed":288,"listed_where_code_ran":73,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/safety-alignment","prev":"/task/safety-alignment","next":"/task/safety-alignment/papers/3","papers":[{"url":"/paper/chatbug-a-common-vulnerability-of-aligned","slug":"chatbug-a-common-vulnerability-of-aligned","title":"ChatBug: A Common Vulnerability of Aligned LLMs Induced by Chat Templates","date":"2024-06-17","arxiv_id":"2406.12935","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbug-a-common-vulnerability-of-aligned#ran","syntology_url":"https://syntology.ai/paper/2406.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12935"}},"official":{"repos":["uw-nsl/ChatBug"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/csrt-evaluation-and-analysis-of-llms-using","slug":"csrt-evaluation-and-analysis-of-llms-using","title":"Code-Switching Red-Teaming: LLM Evaluation for Safety and Multilingual Understanding","date":"2024-06-17","arxiv_id":"2406.15481","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/csrt-evaluation-and-analysis-of-llms-using#ran","syntology_url":"https://syntology.ai/paper/2406.15481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15481"}},"official":{"repos":["haneul-yoo/csrt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-arithmetic-a-framework-for-test-time","slug":"safety-arithmetic-a-framework-for-test-time","title":"Safety Arithmetic: A Framework for Test-time Safety Alignment of Language Models by Steering Parameters and Activations","date":"2024-06-17","arxiv_id":"2406.11801","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-arithmetic-a-framework-for-test-time#ran","syntology_url":"https://syntology.ai/paper/2406.11801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11801"}},"official":{"repos":["declare-lab/safety-arithmetic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-vl-a-comprehensive-safety-preference","slug":"spa-vl-a-comprehensive-safety-preference","title":"SPA-VL: A Comprehensive Safety Preference Alignment Dataset for Vision Language Model","date":"2024-06-17","arxiv_id":"2406.12030","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spa-vl-a-comprehensive-safety-preference#ran","syntology_url":"https://syntology.ai/paper/2406.12030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12030"}},"official":{"repos":["echosechen/spa-vl-rlhf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-alignment-should-be-made-more-than","slug":"safety-alignment-should-be-made-more-than","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","date":"2024-06-10","arxiv_id":"2406.05946","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/safety-alignment-should-be-made-more-than#ran","syntology_url":"https://syntology.ai/paper/2406.05946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05946"}},"official":{"repos":["unispac/shallow-vs-deep-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/how-alignment-and-jailbreak-work-explain-llm","slug":"how-alignment-and-jailbreak-work-explain-llm","title":"How Alignment and Jailbreak Work: Explain LLM Safety through Intermediate Hidden States","date":"2024-06-09","arxiv_id":"2406.05644","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-alignment-and-jailbreak-work-explain-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05644"}},"official":{"repos":["ydyjya/llm-ihs-explanation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/or-bench-an-over-refusal-benchmark-for-large","slug":"or-bench-an-over-refusal-benchmark-for-large","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","date":"2024-05-31","arxiv_id":"2405.20947","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/or-bench-an-over-refusal-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2405.20947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20947"}},"official":{"repos":["justincui03/or-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/one-shot-safety-alignment-for-large-language","slug":"one-shot-safety-alignment-for-large-language","title":"One-Shot Safety Alignment for Large Language Models via Optimal Dualization","date":"2024-05-29","arxiv_id":"2405.19544","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/one-shot-safety-alignment-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.19544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19544"}},"official":{"repos":["shuoli90/CAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lazy-safety-alignment-for-large-language","slug":"lazy-safety-alignment-for-large-language","title":"Lisa: Lazy Safety Alignment for Large Language Models against Harmful Fine-tuning Attack","date":"2024-05-28","arxiv_id":"2405.18641","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-the-safety-landscape-measuring","slug":"navigating-the-safety-landscape-measuring","title":"Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models","date":"2024-05-27","arxiv_id":"2405.17374","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/navigating-the-safety-landscape-measuring#ran","syntology_url":"https://syntology.ai/paper/2405.17374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17374"}},"official":{"repos":["shengyun-peng/llm-landscape"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/parden-can-you-repeat-that-defending-against","slug":"parden-can-you-repeat-that-defending-against","title":"PARDEN, Can You Repeat That? Defending against Jailbreaks via Repetition","date":"2024-05-13","arxiv_id":"2405.07932","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parden-can-you-repeat-that-defending-against#ran","syntology_url":"https://syntology.ai/paper/2405.07932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07932"}},"official":{"repos":["ed-zh/parden"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing","slug":"don-t-say-no-jailbreaking-llm-by-suppressing","title":"Don't Say No: Jailbreaking LLM by Suppressing Refusal","date":"2024-04-25","arxiv_id":"2404.16369","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing#ran","syntology_url":"https://syntology.ai/paper/2404.16369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16369"}},"official":{"repos":["dsn-2024/dsn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-safety-risks-in-open-source-llms","slug":"uncovering-safety-risks-in-open-source-llms","title":"Uncovering Safety Risks of Large Language Models through Concept Activation Vector","date":"2024-04-18","arxiv_id":"2404.12038","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/uncovering-safety-risks-in-open-source-llms#ran","syntology_url":"https://syntology.ai/paper/2404.12038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12038"}},"official":{"repos":["sproutnan/ai-safety_scav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/amplegcg-learning-a-universal-and","slug":"amplegcg-learning-a-universal-and","title":"AmpleGCG: Learning a Universal and Transferable Generative Model of Adversarial Suffixes for Jailbreaking Both Open and Closed LLMs","date":"2024-04-11","arxiv_id":"2404.07921","repositories_listed":1,"syntology":null},{"url":"/paper/eraser-jailbreaking-defense-in-large-language","slug":"eraser-jailbreaking-defense-in-large-language","title":"Eraser: Jailbreaking Defense in Large Language Models via Unlearning Harmful Knowledge","date":"2024-04-08","arxiv_id":"2404.05880","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/eraser-jailbreaking-defense-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2404.05880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05880"}},"official":{"repos":["ZeroNLP/Eraser"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/keeping-llms-aligned-after-fine-tuning-the#ran","syntology_url":"https://syntology.ai/paper/2402.18540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18540"}},"official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/drattack-prompt-decomposition-and","slug":"drattack-prompt-decomposition-and","title":"DrAttack: Prompt Decomposition and Reconstruction Makes Powerful LLM Jailbreakers","date":"2024-02-25","arxiv_id":"2402.16914","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drattack-prompt-decomposition-and#ran","syntology_url":"https://syntology.ai/paper/2402.16914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16914"}},"official":{"repos":["xirui-li/drattack"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-fine-tuning-jailbreak-attack-with","slug":"mitigating-fine-tuning-jailbreak-attack-with","title":"Mitigating Fine-tuning based Jailbreak Attack with Backdoor Enhanced Safety Alignment","date":"2024-02-22","arxiv_id":"2402.14968","repositories_listed":1,"syntology":null},{"url":"/paper/self-distillation-bridges-distribution-gap-in","slug":"self-distillation-bridges-distribution-gap-in","title":"Self-Distillation Bridges Distribution Gap in Language Model Fine-Tuning","date":"2024-02-21","arxiv_id":"2402.13669","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-distillation-bridges-distribution-gap-in#ran","syntology_url":"https://syntology.ai/paper/2402.13669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13669"}},"official":{"repos":["sail-sg/sdft"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/artprompt-ascii-art-based-jailbreak-attacks","slug":"artprompt-ascii-art-based-jailbreak-attacks","title":"ArtPrompt: ASCII Art-based Jailbreak Attacks against Aligned LLMs","date":"2024-02-19","arxiv_id":"2402.11753","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/artprompt-ascii-art-based-jailbreak-attacks#ran","syntology_url":"https://syntology.ai/paper/2402.11753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11753"}},"official":{"repos":["uw-nsl/ArtPrompt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emulated-disalignment-safety-alignment-for","slug":"emulated-disalignment-safety-alignment-for","title":"Emulated Disalignment: Safety Alignment for Large Language Models May Backfire!","date":"2024-02-19","arxiv_id":"2402.12343","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/emulated-disalignment-safety-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2402.12343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12343"}},"official":{"repos":["ZHZisZZ/emulated-disalignment"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/soft-prompt-threats-attacking-safety","slug":"soft-prompt-threats-attacking-safety","title":"Soft Prompt Threats: Attacking Safety Alignment and Unlearning in Open-Source LLMs through the Embedding Space","date":"2024-02-14","arxiv_id":"2402.09063","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soft-prompt-threats-attacking-safety#ran","syntology_url":"https://syntology.ai/paper/2402.09063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09063"}},"official":{"repos":["schwinnl/llm_embedding_attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-fine-tuning-at-almost-no-cost-a","slug":"safety-fine-tuning-at-almost-no-cost-a","title":"Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models","date":"2024-02-03","arxiv_id":"2402.02207","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safety-fine-tuning-at-almost-no-cost-a#ran","syntology_url":"https://syntology.ai/paper/2402.02207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02207"}},"official":{"repos":["ys-zong/vlguard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mllm-protector-ensuring-mllm-s-safety-without","slug":"mllm-protector-ensuring-mllm-s-safety-without","title":"MLLM-Protector: Ensuring MLLM's Safety without Hurting Performance","date":"2024-01-05","arxiv_id":"2401.02906","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mllm-protector-ensuring-mllm-s-safety-without#ran","syntology_url":"https://syntology.ai/paper/2401.02906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02906"}},"official":{"repos":["pipilurj/mllm-protector"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned","slug":"safety-alignment-in-nlp-tasks-weakly-aligned","title":"Safety Alignment in NLP Tasks: Weakly Aligned Summarization as an In-Context Attack","date":"2023-12-12","arxiv_id":"2312.06924","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned#ran","syntology_url":"https://syntology.ai/paper/2312.06924","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06924"}},"official":{"repos":["fyyfu/safetyalignnlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backdoor-activation-attack-attack-large","slug":"backdoor-activation-attack-attack-large","title":"Trojan Activation Attack: Red-Teaming Large Language Models using Activation Steering for Safety-Alignment","date":"2023-11-15","arxiv_id":"2311.09433","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/backdoor-activation-attack-attack-large#ran","syntology_url":"https://syntology.ai/paper/2311.09433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09433"}},"official":{"repos":["wang2226/backdoor-activation-attack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/how-trustworthy-are-open-source-llms-an","slug":"how-trustworthy-are-open-source-llms-an","title":"How Trustworthy are Open-Source LLMs? An Assessment under Malicious Demonstrations Shows their Vulnerabilities","date":"2023-11-15","arxiv_id":"2311.09447","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/how-trustworthy-are-open-source-llms-an#ran","syntology_url":"https://syntology.ai/paper/2311.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09447"}},"official":{"repos":["osu-nlp-group/eval-llm-trust"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/superhf-supervised-iterative-learning-from","slug":"superhf-supervised-iterative-learning-from","title":"SuperHF: Supervised Iterative Learning from Human Feedback","date":"2023-10-25","arxiv_id":"2310.16763","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/superhf-supervised-iterative-learning-from#ran","syntology_url":"https://syntology.ai/paper/2310.16763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16763"}},"official":{"repos":["openfeedback/superhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/autodan-automatic-and-interpretable","slug":"autodan-automatic-and-interpretable","title":"AutoDAN: Interpretable Gradient-Based Adversarial Attacks on Large Language Models","date":"2023-10-23","arxiv_id":"2310.15140","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autodan-automatic-and-interpretable#ran","syntology_url":"https://syntology.ai/paper/2310.15140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15140"}},"official":null}},{"url":"/paper/fine-tuning-aligned-language-models","slug":"fine-tuning-aligned-language-models","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","date":"2023-10-05","arxiv_id":"2310.03693","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fine-tuning-aligned-language-models#ran","syntology_url":"https://syntology.ai/paper/2310.03693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03693"}},"official":{"repos":["llm-tuning-safety/llms-finetuning-safety"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/all-languages-matter-on-the-multilingual","slug":"all-languages-matter-on-the-multilingual","title":"All Languages Matter: On the Multilingual Safety of Large Language Models","date":"2023-10-02","arxiv_id":"2310.00905","repositories_listed":1,"syntology":null},{"url":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/who-is-chatgpt-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2310.01386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01386"}},"official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat","slug":"gpt-4-is-too-smart-to-be-safe-stealthy-chat","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","date":"2023-08-12","arxiv_id":"2308.06463","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat#ran","syntology_url":"https://syntology.ai/paper/2308.06463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06463"}},"official":{"repos":["robustnlp/cipherchat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beavertails-towards-improved-safety-alignment-1","slug":"beavertails-towards-improved-safety-alignment-1","title":"BeaverTails: Towards Improved Safety Alignment of LLM via a Human-Preference Dataset","date":"2023-07-10","arxiv_id":"2307.04657","repositories_listed":1,"syntology":null},{"url":null,"slug":"tuneshield-mitigating-toxicity-in","title":"TuneShield: Mitigating Toxicity in Conversational AI while Fine-tuning on Untrusted Data","date":"2025-07-08","arxiv_id":"2507.05660","repositories_listed":0,"syntology":null},{"url":null,"slug":"trojan-horse-prompting-jailbreaking","title":"Trojan Horse Prompting: Jailbreaking Conversational Multimodal Models by Forging Assistant Message","date":"2025-07-07","arxiv_id":"2507.04673","repositories_listed":0,"syntology":null},{"url":null,"slug":"just-enough-shifts-mitigating-over-refusal-in","title":"Just Enough Shifts: Mitigating Over-Refusal in Aligned Language Models with Targeted Representation Fine-Tuning","date":"2025-07-06","arxiv_id":"2507.04250","repositories_listed":0,"syntology":null},{"url":null,"slug":"security-assessment-of-deepseek-and-gpt","title":"Security Assessment of DeepSeek and GPT Series Models against Jailbreak Attacks","date":"2025-06-23","arxiv_id":"2506.18543","repositories_listed":0,"syntology":null},{"url":null,"slug":"safex-analyzing-vulnerabilities-of-moe-based","title":"SAFEx: Analyzing Vulnerabilities of MoE-Based LLMs via Stable Safety-critical Expert Identification","date":"2025-06-20","arxiv_id":"2506.17368","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-make-it-up-preserving-ignorance","title":"Don't Make It Up: Preserving Ignorance Awareness in LLM Fine-Tuning","date":"2025-06-17","arxiv_id":"2506.14387","repositories_listed":0,"syntology":null},{"url":null,"slug":"securitylingua-efficient-defense-of-llm","title":"SecurityLingua: Efficient Defense of LLM Jailbreak Attacks via Security-Aware Prompt Compression","date":"2025-06-15","arxiv_id":"2506.12707","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-judgment-to-interference-early-stopping","title":"From Judgment to Interference: Early Stopping LLM Harmful Outputs via Streaming Content Monitoring","date":"2025-06-11","arxiv_id":"2506.09996","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attack-safety-alignment-alkali","title":"AdversariaL attacK sAfety aLIgnment(ALKALI): Safeguarding LLMs through GRACE: Geometric Representation-Aware Contrastive Enhancement- Introducing Adversarial Vulnerability Quality Index (AVQI)","date":"2025-06-10","arxiv_id":"2506.08885","repositories_listed":0,"syntology":null},{"url":null,"slug":"refusal-feature-guided-teacher-for-safe","title":"Refusal-Feature-guided Teacher for Safe Finetuning via Data Filtering and Alignment Distillation","date":"2025-06-09","arxiv_id":"2506.07356","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10020","title":"From Threat to Tool: Leveraging Refusal-Aware Injection Attacks for Safety Alignment","date":"2025-06-07","arxiv_id":"2506.10020","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-llm-safety-guardrails-collapse-after-fine","title":"Why LLM Safety Guardrails Collapse After Fine-tuning: A Similarity Analysis Between Alignment and Fine-tuning Datasets","date":"2025-06-05","arxiv_id":"2506.05346","repositories_listed":0,"syntology":null},{"url":"/paper/vulnerability-aware-alignment-mitigating","slug":"vulnerability-aware-alignment-mitigating","title":"Vulnerability-Aware Alignment: Mitigating Uneven Forgetting in Harmful Fine-Tuning","date":"2025-06-04","arxiv_id":"2506.03850","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vulnerability-aware-alignment-mitigating#ran","syntology_url":"https://syntology.ai/paper/2506.03850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03850"}},"official":null}},{"url":null,"slug":"align-is-not-enough-multimodal-universal","title":"Align is not Enough: Multimodal Universal Jailbreak Attack against Multimodal Large Language Models","date":"2025-06-02","arxiv_id":"2506.01307","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-llm-robustness-for-vlm-safety","title":"Bootstrapping LLM Robustness for VLM Safety via Reducing the Pretraining Modality Gap","date":"2025-05-30","arxiv_id":"2505.24208","repositories_listed":0,"syntology":null},{"url":"/paper/evorefuse-evolutionary-prompt-optimization","slug":"evorefuse-evolutionary-prompt-optimization","title":"EVOREFUSE: Evolutionary Prompt Optimization for Evaluation and Mitigation of LLM Over-Refusal to Pseudo-Malicious Instructions","date":"2025-05-29","arxiv_id":"2505.23473","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evorefuse-evolutionary-prompt-optimization#ran","syntology_url":"https://syntology.ai/paper/2505.23473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23473"}},"official":null}},{"url":null,"slug":"safecomm-what-about-safety-alignment-in-fine","title":"SafeCOMM: What about Safety Alignment in Fine-Tuned Telecom Large Language Models?","date":"2025-05-29","arxiv_id":"2506.00062","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-the-threat-vulnerabilities-in-vision","title":"Seeing the Threat: Vulnerabilities in Vision-Language Models to Adversarial Attack","date":"2025-05-28","arxiv_id":"2505.21967","repositories_listed":0,"syntology":null},{"url":null,"slug":"poisonswarm-universal-harmful-information","title":"PoisonSwarm: Universal Harmful Information Synthesis via Model Crowdsourcing","date":"2025-05-27","arxiv_id":"2505.21184","repositories_listed":0,"syntology":null},{"url":null,"slug":"sosbench-benchmarking-safety-alignment-on","title":"SOSBENCH: Benchmarking Safety Alignment on Scientific Knowledge","date":"2025-05-27","arxiv_id":"2505.21605","repositories_listed":0,"syntology":null},{"url":null,"slug":"reshaping-representation-space-to-balance-the","title":"Reshaping Representation Space to Balance the Safety and Over-rejection in Large Audio Language Models","date":"2025-05-26","arxiv_id":"2505.19670","repositories_listed":0,"syntology":null},{"url":null,"slug":"safedpo-a-simple-approach-to-direct","title":"SafeDPO: A Simple Approach to Direct Preference Optimization with Enhanced Safety","date":"2025-05-26","arxiv_id":"2505.20065","repositories_listed":0,"syntology":null},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":null,"slug":"safety-alignment-via-constrained-knowledge","title":"Safety Alignment via Constrained Knowledge Unlearning","date":"2025-05-24","arxiv_id":"2505.18588","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-and-safety-of-diffusion-models-via","title":"Alignment and Safety of Diffusion Models via Reinforcement Learning and Reward Modeling: A Survey","date":"2025-05-23","arxiv_id":"2505.17352","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-mitigating-overrefusal-in","title":"Understanding and Mitigating Overrefusal in LLMs from an Unveiling Perspective of Safety Decision Boundary","date":"2025-05-23","arxiv_id":"2505.18325","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrap-embedding-collapse-trap-to-safeguard","title":"CTRAP: Embedding Collapse Trap to Safeguard Large Language Models from Harmful Fine-Tuning","date":"2025-05-22","arxiv_id":"2505.16559","repositories_listed":0,"syntology":null},{"url":"/paper/from-evaluation-to-defense-advancing-safety","slug":"from-evaluation-to-defense-advancing-safety","title":"From Evaluation to Defense: Advancing Safety in Video Large Language Models","date":"2025-05-22","arxiv_id":"2505.16643","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-evaluation-to-defense-advancing-safety#ran","syntology_url":"https://syntology.ai/paper/2505.16643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16643"}},"official":null}},{"url":"/paper/shape-it-up-restoring-llm-safety-during","slug":"shape-it-up-restoring-llm-safety-during","title":"Shape it Up! Restoring LLM Safety during Finetuning","date":"2025-05-22","arxiv_id":"2505.17196","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shape-it-up-restoring-llm-safety-during#ran","syntology_url":"https://syntology.ai/paper/2505.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17196"}},"official":null}},{"url":null,"slug":"haet-bhasha-aur-diskrimineshun-phonetic","title":"\"Haet Bhasha aur Diskrimineshun\": Phonetic Perturbations in Code-Mixed Hinglish to Red-Team LLMs","date":"2025-05-20","arxiv_id":"2505.14226","repositories_listed":0,"syntology":null},{"url":null,"slug":"safepath-preventing-harmful-reasoning-in","title":"SAFEPATH: Preventing Harmful Reasoning in Chain-of-Thought via Early Alignment","date":"2025-05-20","arxiv_id":"2505.14667","repositories_listed":0,"syntology":null},{"url":null,"slug":"sudollm-on-multi-role-alignment-of-language","title":"sudoLLM : On Multi-role Alignment of Language Models","date":"2025-05-20","arxiv_id":"2505.14607","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-alignment-can-be-not-superficial-with","title":"Safety Alignment Can Be Not Superficial With Explicit Safety Signals","date":"2025-05-19","arxiv_id":"2505.17072","repositories_listed":0,"syntology":null},{"url":"/paper/juli-jailbreak-large-language-models-by-self","slug":"juli-jailbreak-large-language-models-by-self","title":"JULI: Jailbreak Large Language Models by Self-Introspection","date":"2025-05-17","arxiv_id":"2505.11790","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/juli-jailbreak-large-language-models-by-self#ran","syntology_url":"https://syntology.ai/paper/2505.11790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11790"}},"official":null}},{"url":null,"slug":"safevid-toward-safety-aligned-video-large","title":"SafeVid: Toward Safety Aligned Video Large Multimodal Models","date":"2025-05-17","arxiv_id":"2505.11926","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11413","title":"CARES: Comprehensive Evaluation of Safety and Adversarial Robustness in Medical LLMs","date":"2025-05-16","arxiv_id":"2505.11413","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-injection-systemically-degrades-large","title":"Noise Injection Systemically Degrades Large Language Model Safety Guardrails","date":"2025-05-16","arxiv_id":"2505.13500","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-safety-risks-in-llms-fine-tuned","title":"Analysing Safety Risks in LLMs Fine-Tuned with Pseudo-Malicious Cyber Security Data","date":"2025-05-15","arxiv_id":"2505.09974","repositories_listed":0,"syntology":null},{"url":null,"slug":"falsereject-a-resource-for-improving","title":"FalseReject: A Resource for Improving Contextual Safety and Mitigating Over-Refusals in LLMs via Structured Reasoning","date":"2025-05-12","arxiv_id":"2505.08054","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-trigger-token-is-enough-a-defense","title":"One Trigger Token Is Enough: A Defense Strategy for Balancing Safety and Usability in Large Language Models","date":"2025-05-12","arxiv_id":"2505.07167","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurel-attack-neuron-relearning-for-safety","title":"NeuRel-Attack: Neuron Relearning for Safety Disalignment in Large Language Models","date":"2025-04-29","arxiv_id":"2504.21053","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-s-pulling-the-strings-evaluating","title":"What's Pulling the Strings? Evaluating Integrity and Attribution in AI Training and Inference through Concept Shift","date":"2025-04-28","arxiv_id":"2504.21042","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-awareness","title":"AI Awareness","date":"2025-04-25","arxiv_id":"2504.20084","repositories_listed":0,"syntology":null},{"url":null,"slug":"aixamine-llm-safety-and-security-simplified","title":"aiXamine: Simplified LLM Safety and Security","date":"2025-04-21","arxiv_id":"2504.14985","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-nsfw-free-text-to-image-generation","title":"Towards NSFW-Free Text-to-Image Generation via Safety-Constraint Direct Preference Optimization","date":"2025-04-19","arxiv_id":"2504.14290","repositories_listed":0,"syntology":null},{"url":null,"slug":"thought-manipulation-external-thought-can-be","title":"Thought Manipulation: External Thought Can Be Efficient for Large Reasoning Models","date":"2025-04-18","arxiv_id":"2504.13626","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmguard-r1-proactive-safety-alignment-for","title":"VLMGuard-R1: Proactive Safety Alignment for VLMs via Reasoning-Driven Prompt Optimization","date":"2025-04-17","arxiv_id":"2504.12661","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-teaming-multi-turn-jailbreaks-and-defenses","title":"X-Teaming: Multi-Turn Jailbreaks and Defenses with Adaptive Multi-Agents","date":"2025-04-15","arxiv_id":"2504.13203","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-we-really-need-curated-malicious-data-for","title":"Do We Really Need Curated Malicious Data for Safety Alignment in Multi-modal Large Language Models?","date":"2025-04-14","arxiv_id":"2504.10000","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-can-be-a-dangerous-persuader-empirical","title":"LLM Can be a Dangerous Persuader: Empirical Study of Persuasion Safety in Large Language Models","date":"2025-04-14","arxiv_id":"2504.10430","repositories_listed":0,"syntology":null},{"url":null,"slug":"realsafe-r1-safety-aligned-deepseek-r1","title":"RealSafe-R1: Safety-Aligned DeepSeek-R1 without Compromising Reasoning Capability","date":"2025-04-14","arxiv_id":"2504.10081","repositories_listed":0,"syntology":null},{"url":null,"slug":"safemlrm-demystifying-safety-in-multi-modal","title":"SafeMLRM: Demystifying Safety in Multi-modal Large Reasoning Models","date":"2025-04-09","arxiv_id":"2504.08813","repositories_listed":0,"syntology":null},{"url":null,"slug":"erpo-advancing-safety-alignment-via-ex-ante","title":"ERPO: Advancing Safety Alignment via Ex-Ante Reasoning Preference Optimization","date":"2025-04-03","arxiv_id":"2504.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-is-less-the-pitfalls-of-multi-model","title":"More is Less: The Pitfalls of Multi-Model Synthetic Preference Data in DPO Safety Alignment","date":"2025-04-03","arxiv_id":"2504.02193","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-1-safer-alignment-of-reasoning-llms-with","title":"STAR-1: Safer Alignment of Reasoning LLMs with 1K Data","date":"2025-04-02","arxiv_id":"2504.01903","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectively-controlling-reasoning-models","title":"Effectively Controlling Reasoning Models through Thinking Intervention","date":"2025-03-31","arxiv_id":"2503.24370","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-rlhf-v-safe-reinforcement-learning-from","title":"Safe RLHF-V: Safe Reinforcement Learning from Human Feedback in Multimodal Large Language Models","date":"2025-03-22","arxiv_id":"2503.17682","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-in-depth-defending-jailbreak-attacks","title":"Align in Depth: Defending Jailbreak Attacks via Progressive Answer Detoxification","date":"2025-03-14","arxiv_id":"2503.11185","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-based-reward-modeling-for","title":"Representation-based Reward Modeling for Efficient Safety Alignment of Large Language Model","date":"2025-03-13","arxiv_id":"2503.10093","repositories_listed":0,"syntology":null},{"url":null,"slug":"jbfuzz-jailbreaking-llms-efficiently-and","title":"JBFuzz: Jailbreaking LLMs Efficiently and Effectively Using Fuzzing","date":"2025-03-12","arxiv_id":"2503.08990","repositories_listed":0,"syntology":null},{"url":null,"slug":"backtracking-for-safety","title":"Backtracking for Safety","date":"2025-03-11","arxiv_id":"2503.08919","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-jailbreak-probability-to-attack-and","title":"Utilizing Jailbreak Probability to Attack and Safeguard Multimodal LLMs","date":"2025-03-10","arxiv_id":"2503.06989","repositories_listed":0,"syntology":null},{"url":null,"slug":"safearena-evaluating-the-safety-of-autonomous","title":"SafeArena: Evaluating the Safety of Autonomous Web Agents","date":"2025-03-06","arxiv_id":"2503.04957","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-is-not-only-about-refusal-reasoning","title":"Safety is Not Only About Refusal: Reasoning-Enhanced Fine-tuning for Interpretable LLM Safety","date":"2025-03-06","arxiv_id":"2503.05021","repositories_listed":0,"syntology":null},{"url":null,"slug":"safevla-towards-safety-alignment-of-vision","title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","date":"2025-03-05","arxiv_id":"2503.03480","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-safety-evaluations-lack-robustness","title":"LLM-Safety Evaluations Lack Robustness","date":"2025-03-04","arxiv_id":"2503.02574","repositories_listed":0,"syntology":null}],"record_sha256":"e572ca7ff0d84c39755ecef6b29f31cd5213eb688f61861bdc9f1ea92616934f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}