{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/safety-alignment/papers/ran/1","list_of":"/task/safety-alignment","task":"Safety Alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,73],"of":73,"counts":{"archive_papers_tagged":288,"with_a_code_link":134,"where_syntology_ran_a_sample":73,"not_listed_spam_title":0,"listed":288,"listed_where_code_ran":73,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/safety-alignment/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/the-devil-behind-the-mask-an-emergent-safety","slug":"the-devil-behind-the-mask-an-emergent-safety","title":"The Devil behind the mask: An emergent safety vulnerability of Diffusion LLMs","date":"2025-07-15","arxiv_id":"2507.11097","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-devil-behind-the-mask-an-emergent-safety#ran","syntology_url":"https://syntology.ai/paper/2507.11097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11097"}},"official":{"repos":["zichenwen1/dija"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-pruning-lora-robust-distance-guided","slug":"safe-pruning-lora-robust-distance-guided","title":"Safe Pruning LoRA: Robust Distance-Guided Pruning for Safety Alignment in Adaptation of LLMs","date":"2025-06-21","arxiv_id":"2506.18931","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-pruning-lora-robust-distance-guided#ran","syntology_url":"https://syntology.ai/paper/2506.18931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18931"}},"official":{"repos":["aoshuang92/splora"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chasing-moving-targets-with-online-self-play","slug":"chasing-moving-targets-with-online-self-play","title":"Chasing Moving Targets with Online Self-Play Reinforcement Learning for Safer Language Models","date":"2025-06-09","arxiv_id":"2506.07468","repositories_listed":1,"syntology":{"n":17,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/chasing-moving-targets-with-online-self-play#ran","syntology_url":"https://syntology.ai/paper/2506.07468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07468"}},"official":{"repos":["mickelliu/selfplay-redteaming"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":11,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vulnerability-aware-alignment-mitigating","slug":"vulnerability-aware-alignment-mitigating","title":"Vulnerability-Aware Alignment: Mitigating Uneven Forgetting in Harmful Fine-Tuning","date":"2025-06-04","arxiv_id":"2506.03850","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vulnerability-aware-alignment-mitigating#ran","syntology_url":"https://syntology.ai/paper/2506.03850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03850"}},"official":null}},{"url":"/paper/diablo-diagonal-blocks-are-sufficient-for","slug":"diablo-diagonal-blocks-are-sufficient-for","title":"DiaBlo: Diagonal Blocks Are Sufficient For Finetuning","date":"2025-06-03","arxiv_id":"2506.03230","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diablo-diagonal-blocks-are-sufficient-for#ran","syntology_url":"https://syntology.ai/paper/2506.03230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03230"}},"official":{"repos":["ziyangjoy/diablo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evorefuse-evolutionary-prompt-optimization","slug":"evorefuse-evolutionary-prompt-optimization","title":"EVOREFUSE: Evolutionary Prompt Optimization for Evaluation and Mitigation of LLM Over-Refusal to Pseudo-Malicious Instructions","date":"2025-05-29","arxiv_id":"2505.23473","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evorefuse-evolutionary-prompt-optimization#ran","syntology_url":"https://syntology.ai/paper/2505.23473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23473"}},"official":null}},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":"/paper/from-evaluation-to-defense-advancing-safety","slug":"from-evaluation-to-defense-advancing-safety","title":"From Evaluation to Defense: Advancing Safety in Video Large Language Models","date":"2025-05-22","arxiv_id":"2505.16643","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-evaluation-to-defense-advancing-safety#ran","syntology_url":"https://syntology.ai/paper/2505.16643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16643"}},"official":null}},{"url":"/paper/mpo-multilingual-safety-alignment-via-reward","slug":"mpo-multilingual-safety-alignment-via-reward","title":"MPO: Multilingual Safety Alignment via Reward Gap Optimization","date":"2025-05-22","arxiv_id":"2505.16869","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mpo-multilingual-safety-alignment-via-reward#ran","syntology_url":"https://syntology.ai/paper/2505.16869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16869"}},"official":{"repos":["circle-hit/mpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shape-it-up-restoring-llm-safety-during","slug":"shape-it-up-restoring-llm-safety-during","title":"Shape it Up! Restoring LLM Safety during Finetuning","date":"2025-05-22","arxiv_id":"2505.17196","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shape-it-up-restoring-llm-safety-during#ran","syntology_url":"https://syntology.ai/paper/2505.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17196"}},"official":null}},{"url":"/paper/safety-subspaces-are-not-distinct-a-fine","slug":"safety-subspaces-are-not-distinct-a-fine","title":"Safety Subspaces are Not Distinct: A Fine-Tuning Case Study","date":"2025-05-20","arxiv_id":"2505.14185","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safety-subspaces-are-not-distinct-a-fine#ran","syntology_url":"https://syntology.ai/paper/2505.14185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14185"}},"official":{"repos":["cert-lab/safety-subspaces"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/linear-control-of-test-awareness-reveals","slug":"linear-control-of-test-awareness-reveals","title":"Linear Control of Test Awareness Reveals Differential Compliance in Reasoning Models","date":"2025-05-20","arxiv_id":"2505.14617","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/linear-control-of-test-awareness-reveals#ran","syntology_url":"https://syntology.ai/paper/2505.14617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14617"}},"official":{"repos":["microsoft/test_awareness_steering"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/juli-jailbreak-large-language-models-by-self","slug":"juli-jailbreak-large-language-models-by-self","title":"JULI: Jailbreak Large Language Models by Self-Introspection","date":"2025-05-17","arxiv_id":"2505.11790","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/juli-jailbreak-large-language-models-by-self#ran","syntology_url":"https://syntology.ai/paper/2505.11790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11790"}},"official":null}},{"url":"/paper/benign-samples-matter-fine-tuning-on-outlier","slug":"benign-samples-matter-fine-tuning-on-outlier","title":"Benign Samples Matter! Fine-tuning On Outlier Benign Samples Severely Breaks Safety","date":"2025-05-11","arxiv_id":"2505.06843","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benign-samples-matter-fine-tuning-on-outlier#ran","syntology_url":"https://syntology.ai/paper/2505.06843","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.06843"}},"official":{"repos":["guanzihan/benign-samples-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sudo-rm-rf-agentic-security","slug":"sudo-rm-rf-agentic-security","title":"sudo rm -rf agentic_security","date":"2025-03-26","arxiv_id":"2503.20279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sudo-rm-rf-agentic-security#ran","syntology_url":"https://syntology.ai/paper/2503.20279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20279"}},"official":{"repos":["AIM-Intelligence/SUDO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safemerge-preserving-safety-alignment-in-fine","slug":"safemerge-preserving-safety-alignment-in-fine","title":"SafeMERGE: Preserving Safety Alignment in Fine-Tuned Large Language Models via Selective Layer-Wise Model Merging","date":"2025-03-21","arxiv_id":"2503.17239","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safemerge-preserving-safety-alignment-in-fine#ran","syntology_url":"https://syntology.ai/paper/2503.17239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.17239"}},"official":{"repos":["aladinD/SafeMERGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-llm-safety-alignment-with-dual","slug":"improving-llm-safety-alignment-with-dual","title":"Improving LLM Safety Alignment with Dual-Objective Optimization","date":"2025-03-05","arxiv_id":"2503.03710","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-llm-safety-alignment-with-dual#ran","syntology_url":"https://syntology.ai/paper/2503.03710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03710"}},"official":{"repos":["wicai24/door-alignment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2503-00555","slug":"2503-00555","title":"Safety Tax: Safety Alignment Makes Your Large Reasoning Models Less Reasonable","date":"2025-03-01","arxiv_id":"2503.00555","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2503-00555#ran","syntology_url":"https://syntology.ai/paper/2503.00555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00555"}},"official":{"repos":["git-disl/safety-tax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-dialogue-dynamics-for-robustness","slug":"steering-dialogue-dynamics-for-robustness","title":"Steering Dialogue Dynamics for Robustness against Multi-turn Jailbreaking Attacks","date":"2025-02-28","arxiv_id":"2503.00187","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/steering-dialogue-dynamics-for-robustness#ran","syntology_url":"https://syntology.ai/paper/2503.00187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00187"}},"official":{"repos":["hanjianghu/nbf-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-boundary-establishing-exact-safety-boundary","slug":"x-boundary-establishing-exact-safety-boundary","title":"X-Boundary: Establishing Exact Safety Boundary to Shield LLMs from Multi-Turn Jailbreaks without Compromising Usability","date":"2025-02-14","arxiv_id":"2502.09990","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x-boundary-establishing-exact-safety-boundary#ran","syntology_url":"https://syntology.ai/paper/2502.09990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09990"}},"official":{"repos":["ai45lab/x-boundary"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/making-them-a-malicious-database-exploiting","slug":"making-them-a-malicious-database-exploiting","title":"QueryAttack: Jailbreaking Aligned Large Language Models Using Structured Non-natural Query Language","date":"2025-02-13","arxiv_id":"2502.09723","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-them-a-malicious-database-exploiting#ran","syntology_url":"https://syntology.ai/paper/2502.09723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09723"}},"official":{"repos":["horizonsinzqs/queryattack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speak-easy-eliciting-harmful-jailbreaks-from","slug":"speak-easy-eliciting-harmful-jailbreaks-from","title":"Speak Easy: Eliciting Harmful Jailbreaks from LLMs with Simple Interactions","date":"2025-02-06","arxiv_id":"2502.04322","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-easy-eliciting-harmful-jailbreaks-from#ran","syntology_url":"https://syntology.ai/paper/2502.04322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04322"}},"official":{"repos":["yiksiu-chan/SpeakEasy"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pandas-improving-many-shot-jailbreaking-via","slug":"pandas-improving-many-shot-jailbreaking-via","title":"PANDAS: Improving Many-shot Jailbreaking via Positive Affirmation, Negative Demonstration, and Adaptive Sampling","date":"2025-02-04","arxiv_id":"2502.01925","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pandas-improving-many-shot-jailbreaking-via#ran","syntology_url":"https://syntology.ai/paper/2502.01925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01925"}},"official":{"repos":["averyma/pandas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stair-improving-safety-alignment-with","slug":"stair-improving-safety-alignment-with","title":"STAIR: Improving Safety Alignment with Introspective Reasoning","date":"2025-02-04","arxiv_id":"2502.02384","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stair-improving-safety-alignment-with#ran","syntology_url":"https://syntology.ai/paper/2502.02384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02384"}},"official":{"repos":["thu-ml/stair"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-safety-alignment-is-divergence-estimation","slug":"llm-safety-alignment-is-divergence-estimation","title":"LLM Safety Alignment is Divergence Estimation in Disguise","date":"2025-02-02","arxiv_id":"2502.00657","repositories_listed":1,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/llm-safety-alignment-is-divergence-estimation#ran","syntology_url":"https://syntology.ai/paper/2502.00657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00657"}},"official":{"repos":["rhaldarpurdue/kldo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/virus-harmful-fine-tuning-attack-for-large","slug":"virus-harmful-fine-tuning-attack-for-large","title":"Virus: Harmful Fine-tuning Attack for Large Language Models Bypassing Guardrail Moderation","date":"2025-01-29","arxiv_id":"2501.17433","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/virus-harmful-fine-tuning-attack-for-large#ran","syntology_url":"https://syntology.ai/paper/2501.17433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17433"}},"official":{"repos":["git-disl/virus"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/separate-the-wheat-from-the-chaff-a-post-hoc","slug":"separate-the-wheat-from-the-chaff-a-post-hoc","title":"Separate the Wheat from the Chaff: A Post-Hoc Approach to Safety Re-Alignment for Fine-Tuned Language Models","date":"2024-12-15","arxiv_id":"2412.11041","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/separate-the-wheat-from-the-chaff-a-post-hoc#ran","syntology_url":"https://syntology.ai/paper/2412.11041","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11041"}},"official":{"repos":["pikepokenew/IRR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/immune-improving-safety-against-jailbreaks-in","slug":"immune-improving-safety-against-jailbreaks-in","title":"Immune: Improving Safety Against Jailbreaks in Multi-modal LLMs via Inference-Time Alignment","date":"2024-11-27","arxiv_id":"2411.18688","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/immune-improving-safety-against-jailbreaks-in#ran","syntology_url":"https://syntology.ai/paper/2411.18688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18688"}},"official":null}},{"url":"/paper/sg-bench-evaluating-llm-safety-generalization","slug":"sg-bench-evaluating-llm-safety-generalization","title":"SG-Bench: Evaluating LLM Safety Generalization Across Diverse Tasks and Prompt Types","date":"2024-10-29","arxiv_id":"2410.21965","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sg-bench-evaluating-llm-safety-generalization#ran","syntology_url":"https://syntology.ai/paper/2410.21965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21965"}},"official":{"repos":["MurrayTom/SG-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-safety-in-reinforcement-learning","slug":"enhancing-safety-in-reinforcement-learning","title":"Enhancing Safety in Reinforcement Learning with Human Feedback via Rectified Policy Optimization","date":"2024-10-25","arxiv_id":"2410.19933","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-safety-in-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2410.19933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19933"}},"official":{"repos":["pxywatermoon/repo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-scaling-laws-for-in-context-learning","slug":"bayesian-scaling-laws-for-in-context-learning","title":"Bayesian scaling laws for in-context learning","date":"2024-10-21","arxiv_id":"2410.16531","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayesian-scaling-laws-for-in-context-learning#ran","syntology_url":"https://syntology.ai/paper/2410.16531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16531"}},"official":{"repos":["aryamanarora/bayesian-laws-icl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-common-pitfall-of-margin-based-language","slug":"a-common-pitfall-of-margin-based-language","title":"A Common Pitfall of Margin-based Language Model Alignment: Gradient Entanglement","date":"2024-10-17","arxiv_id":"2410.13828","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-common-pitfall-of-margin-based-language#ran","syntology_url":"https://syntology.ai/paper/2410.13828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13828"}},"official":{"repos":["humainlab/understand_marginpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/locking-down-the-finetuned-llms-safety","slug":"locking-down-the-finetuned-llms-safety","title":"Locking Down the Finetuned LLMs Safety","date":"2024-10-14","arxiv_id":"2410.10343","repositories_listed":1,"syntology":{"n":21,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":21,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/locking-down-the-finetuned-llms-safety#ran","syntology_url":"https://syntology.ai/paper/2410.10343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10343"}},"official":{"repos":["zhu-minjun/safetylock"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/derail-yourself-multi-turn-llm-jailbreak","slug":"derail-yourself-multi-turn-llm-jailbreak","title":"Derail Yourself: Multi-turn LLM Jailbreak Attack through Self-discovered Clues","date":"2024-10-14","arxiv_id":"2410.10700","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/derail-yourself-multi-turn-llm-jailbreak#ran","syntology_url":"https://syntology.ai/paper/2410.10700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10700"}},"official":{"repos":["renqibing/actorattack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/targeted-vaccine-safety-alignment-for-large","slug":"targeted-vaccine-safety-alignment-for-large","title":"Targeted Vaccine: Safety Alignment for Large Language Models against Harmful Fine-Tuning via Layer-wise Perturbation","date":"2024-10-13","arxiv_id":"2410.09760","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":4,"n_instrument":8,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 8 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/targeted-vaccine-safety-alignment-for-large#ran","syntology_url":"https://syntology.ai/paper/2410.09760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09760"}},"official":{"repos":["lslland/t-vaccine"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-advisor-dynamic-data-curation-for-safety","slug":"data-advisor-dynamic-data-curation-for-safety","title":"Data Advisor: Dynamic Data Curation for Safety Alignment of Large Language Models","date":"2024-10-07","arxiv_id":"2410.05269","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-advisor-dynamic-data-curation-for-safety#ran","syntology_url":"https://syntology.ai/paper/2410.05269","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05269"}},"official":{"repos":["feiwang96/Data-Advisor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harmful-fine-tuning-attacks-and-defenses-for","slug":"harmful-fine-tuning-attacks-and-defenses-for","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","date":"2024-09-26","arxiv_id":"2409.18169","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/harmful-fine-tuning-attacks-and-defenses-for#ran","syntology_url":"https://syntology.ai/paper/2409.18169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18169"}},"official":{"repos":["git-disl/awesome_llm-harmful-fine-tuning-papers"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/unlocking-adversarial-suffix-optimization","slug":"unlocking-adversarial-suffix-optimization","title":"Unlocking Adversarial Suffix Optimization Without Affirmative Phrases: Efficient Black-box Jailbreaking via LLM as Optimizer","date":"2024-08-21","arxiv_id":"2408.11313","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlocking-adversarial-suffix-optimization#ran","syntology_url":"https://syntology.ai/paper/2408.11313","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11313"}},"official":{"repos":["lenijwp/eclipse"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nothing-in-excess-mitigating-the-exaggerated","slug":"nothing-in-excess-mitigating-the-exaggerated","title":"SCANS: Mitigating the Exaggerated Safety for LLMs via Safety-Conscious Activation Steering","date":"2024-08-21","arxiv_id":"2408.11491","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nothing-in-excess-mitigating-the-exaggerated#ran","syntology_url":"https://syntology.ai/paper/2408.11491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11491"}},"official":{"repos":["zouyingcao/SCANS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/antidote-post-fine-tuning-safety-alignment","slug":"antidote-post-fine-tuning-safety-alignment","title":"Antidote: Post-fine-tuning Safety Alignment for Large Language Models against Harmful Fine-tuning","date":"2024-08-18","arxiv_id":"2408.09600","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/antidote-post-fine-tuning-safety-alignment#ran","syntology_url":"https://syntology.ai/paper/2408.09600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09600"}},"official":null}},{"url":"/paper/2407-21659","slug":"2407-21659","title":"Cross-modality Information Check for Detecting Jailbreaking in Multimodal Large Language Models","date":"2024-07-31","arxiv_id":"2407.21659","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2407-21659#ran","syntology_url":"https://syntology.ai/paper/2407.21659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21659"}},"official":{"repos":["pandragonxiii/cider"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-editing-llms-inject-harm","slug":"can-editing-llms-inject-harm","title":"Can Editing LLMs Inject Harm?","date":"2024-07-29","arxiv_id":"2407.20224","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-editing-llms-inject-harm#ran","syntology_url":"https://syntology.ai/paper/2407.20224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20224"}},"official":null}},{"url":"/paper/q-adapter-training-your-llm-adapter-as-a","slug":"q-adapter-training-your-llm-adapter-as-a","title":"Q-Adapter: Customizing Pre-trained LLMs to New Preferences with Forgetting Mitigation","date":"2024-07-04","arxiv_id":"2407.03856","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/q-adapter-training-your-llm-adapter-as-a#ran","syntology_url":"https://syntology.ai/paper/2407.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03856"}},"official":{"repos":["mansicer/Q-Adapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/safe-unlearning-a-surprisingly-effective-and","slug":"safe-unlearning-a-surprisingly-effective-and","title":"From Theft to Bomb-Making: The Ripple Effect of Unlearning in Defending Against Jailbreak Attacks","date":"2024-07-03","arxiv_id":"2407.02855","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/safe-unlearning-a-surprisingly-effective-and#ran","syntology_url":"https://syntology.ai/paper/2407.02855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02855"}},"official":{"repos":["thu-coai/safeunlearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/sop-unlock-the-power-of-social-facilitation","slug":"sop-unlock-the-power-of-social-facilitation","title":"SeqAR: Jailbreak LLMs with Sequential Auto-Generated Characters","date":"2024-07-02","arxiv_id":"2407.01902","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sop-unlock-the-power-of-social-facilitation#ran","syntology_url":"https://syntology.ai/paper/2407.01902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01902"}},"official":{"repos":["yang-yan-yang-yan/sop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modality-safety-alignment","slug":"cross-modality-safety-alignment","title":"Cross-Modality Safety Alignment","date":"2024-06-21","arxiv_id":"2406.15279","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modality-safety-alignment#ran","syntology_url":"https://syntology.ai/paper/2406.15279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15279"}},"official":{"repos":["sinwang20/siuo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/finding-safety-neurons-in-large-language","slug":"finding-safety-neurons-in-large-language","title":"Finding Safety Neurons in Large Language Models","date":"2024-06-20","arxiv_id":"2406.14144","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finding-safety-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.14144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14144"}},"official":null}},{"url":"/paper/safety-arithmetic-a-framework-for-test-time","slug":"safety-arithmetic-a-framework-for-test-time","title":"Safety Arithmetic: A Framework for Test-time Safety Alignment of Language Models by Steering Parameters and Activations","date":"2024-06-17","arxiv_id":"2406.11801","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-arithmetic-a-framework-for-test-time#ran","syntology_url":"https://syntology.ai/paper/2406.11801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11801"}},"official":{"repos":["declare-lab/safety-arithmetic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-vl-a-comprehensive-safety-preference","slug":"spa-vl-a-comprehensive-safety-preference","title":"SPA-VL: A Comprehensive Safety Preference Alignment Dataset for Vision Language Model","date":"2024-06-17","arxiv_id":"2406.12030","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spa-vl-a-comprehensive-safety-preference#ran","syntology_url":"https://syntology.ai/paper/2406.12030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12030"}},"official":{"repos":["echosechen/spa-vl-rlhf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chatbug-a-common-vulnerability-of-aligned","slug":"chatbug-a-common-vulnerability-of-aligned","title":"ChatBug: A Common Vulnerability of Aligned LLMs Induced by Chat Templates","date":"2024-06-17","arxiv_id":"2406.12935","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbug-a-common-vulnerability-of-aligned#ran","syntology_url":"https://syntology.ai/paper/2406.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12935"}},"official":{"repos":["uw-nsl/ChatBug"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/csrt-evaluation-and-analysis-of-llms-using","slug":"csrt-evaluation-and-analysis-of-llms-using","title":"Code-Switching Red-Teaming: LLM Evaluation for Safety and Multilingual Understanding","date":"2024-06-17","arxiv_id":"2406.15481","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/csrt-evaluation-and-analysis-of-llms-using#ran","syntology_url":"https://syntology.ai/paper/2406.15481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15481"}},"official":{"repos":["haneul-yoo/csrt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-alignment-should-be-made-more-than","slug":"safety-alignment-should-be-made-more-than","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","date":"2024-06-10","arxiv_id":"2406.05946","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/safety-alignment-should-be-made-more-than#ran","syntology_url":"https://syntology.ai/paper/2406.05946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05946"}},"official":{"repos":["unispac/shallow-vs-deep-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/how-alignment-and-jailbreak-work-explain-llm","slug":"how-alignment-and-jailbreak-work-explain-llm","title":"How Alignment and Jailbreak Work: Explain LLM Safety through Intermediate Hidden States","date":"2024-06-09","arxiv_id":"2406.05644","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-alignment-and-jailbreak-work-explain-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05644"}},"official":{"repos":["ydyjya/llm-ihs-explanation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/or-bench-an-over-refusal-benchmark-for-large","slug":"or-bench-an-over-refusal-benchmark-for-large","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","date":"2024-05-31","arxiv_id":"2405.20947","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/or-bench-an-over-refusal-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2405.20947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20947"}},"official":{"repos":["justincui03/or-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/one-shot-safety-alignment-for-large-language","slug":"one-shot-safety-alignment-for-large-language","title":"One-Shot Safety Alignment for Large Language Models via Optimal Dualization","date":"2024-05-29","arxiv_id":"2405.19544","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/one-shot-safety-alignment-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.19544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19544"}},"official":{"repos":["shuoli90/CAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/navigating-the-safety-landscape-measuring","slug":"navigating-the-safety-landscape-measuring","title":"Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models","date":"2024-05-27","arxiv_id":"2405.17374","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/navigating-the-safety-landscape-measuring#ran","syntology_url":"https://syntology.ai/paper/2405.17374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17374"}},"official":{"repos":["shengyun-peng/llm-landscape"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/parden-can-you-repeat-that-defending-against","slug":"parden-can-you-repeat-that-defending-against","title":"PARDEN, Can You Repeat That? Defending against Jailbreaks via Repetition","date":"2024-05-13","arxiv_id":"2405.07932","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parden-can-you-repeat-that-defending-against#ran","syntology_url":"https://syntology.ai/paper/2405.07932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07932"}},"official":{"repos":["ed-zh/parden"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing","slug":"don-t-say-no-jailbreaking-llm-by-suppressing","title":"Don't Say No: Jailbreaking LLM by Suppressing Refusal","date":"2024-04-25","arxiv_id":"2404.16369","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing#ran","syntology_url":"https://syntology.ai/paper/2404.16369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16369"}},"official":{"repos":["dsn-2024/dsn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-safety-risks-in-open-source-llms","slug":"uncovering-safety-risks-in-open-source-llms","title":"Uncovering Safety Risks of Large Language Models through Concept Activation Vector","date":"2024-04-18","arxiv_id":"2404.12038","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/uncovering-safety-risks-in-open-source-llms#ran","syntology_url":"https://syntology.ai/paper/2404.12038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12038"}},"official":{"repos":["sproutnan/ai-safety_scav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/exploring-safety-generalization-challenges-of","slug":"exploring-safety-generalization-challenges-of","title":"CodeAttack: Revealing Safety Generalization Challenges of Large Language Models via Code Completion","date":"2024-03-12","arxiv_id":"2403.07865","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-safety-generalization-challenges-of#ran","syntology_url":"https://syntology.ai/paper/2403.07865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07865"}},"official":{"repos":["renqibing/CodeAttack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/keeping-llms-aligned-after-fine-tuning-the#ran","syntology_url":"https://syntology.ai/paper/2402.18540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18540"}},"official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/drattack-prompt-decomposition-and","slug":"drattack-prompt-decomposition-and","title":"DrAttack: Prompt Decomposition and Reconstruction Makes Powerful LLM Jailbreakers","date":"2024-02-25","arxiv_id":"2402.16914","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drattack-prompt-decomposition-and#ran","syntology_url":"https://syntology.ai/paper/2402.16914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16914"}},"official":{"repos":["xirui-li/drattack"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/self-distillation-bridges-distribution-gap-in","slug":"self-distillation-bridges-distribution-gap-in","title":"Self-Distillation Bridges Distribution Gap in Language Model Fine-Tuning","date":"2024-02-21","arxiv_id":"2402.13669","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-distillation-bridges-distribution-gap-in#ran","syntology_url":"https://syntology.ai/paper/2402.13669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13669"}},"official":{"repos":["sail-sg/sdft"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/artprompt-ascii-art-based-jailbreak-attacks","slug":"artprompt-ascii-art-based-jailbreak-attacks","title":"ArtPrompt: ASCII Art-based Jailbreak Attacks against Aligned LLMs","date":"2024-02-19","arxiv_id":"2402.11753","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/artprompt-ascii-art-based-jailbreak-attacks#ran","syntology_url":"https://syntology.ai/paper/2402.11753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11753"}},"official":{"repos":["uw-nsl/ArtPrompt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-prompt-threats-attacking-safety","slug":"soft-prompt-threats-attacking-safety","title":"Soft Prompt Threats: Attacking Safety Alignment and Unlearning in Open-Source LLMs through the Embedding Space","date":"2024-02-14","arxiv_id":"2402.09063","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soft-prompt-threats-attacking-safety#ran","syntology_url":"https://syntology.ai/paper/2402.09063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09063"}},"official":{"repos":["schwinnl/llm_embedding_attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-fine-tuning-at-almost-no-cost-a","slug":"safety-fine-tuning-at-almost-no-cost-a","title":"Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models","date":"2024-02-03","arxiv_id":"2402.02207","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safety-fine-tuning-at-almost-no-cost-a#ran","syntology_url":"https://syntology.ai/paper/2402.02207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02207"}},"official":{"repos":["ys-zong/vlguard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mllm-protector-ensuring-mllm-s-safety-without","slug":"mllm-protector-ensuring-mllm-s-safety-without","title":"MLLM-Protector: Ensuring MLLM's Safety without Hurting Performance","date":"2024-01-05","arxiv_id":"2401.02906","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mllm-protector-ensuring-mllm-s-safety-without#ran","syntology_url":"https://syntology.ai/paper/2401.02906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02906"}},"official":{"repos":["pipilurj/mllm-protector"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned","slug":"safety-alignment-in-nlp-tasks-weakly-aligned","title":"Safety Alignment in NLP Tasks: Weakly Aligned Summarization as an In-Context Attack","date":"2023-12-12","arxiv_id":"2312.06924","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned#ran","syntology_url":"https://syntology.ai/paper/2312.06924","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06924"}},"official":{"repos":["fyyfu/safetyalignnlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backdoor-activation-attack-attack-large","slug":"backdoor-activation-attack-attack-large","title":"Trojan Activation Attack: Red-Teaming Large Language Models using Activation Steering for Safety-Alignment","date":"2023-11-15","arxiv_id":"2311.09433","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/backdoor-activation-attack-attack-large#ran","syntology_url":"https://syntology.ai/paper/2311.09433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09433"}},"official":{"repos":["wang2226/backdoor-activation-attack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/figstep-jailbreaking-large-vision-language","slug":"figstep-jailbreaking-large-vision-language","title":"FigStep: Jailbreaking Large Vision-Language Models via Typographic Visual Prompts","date":"2023-11-09","arxiv_id":"2311.05608","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/figstep-jailbreaking-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2311.05608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05608"}},"official":{"repos":["thuccslab/figstep"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/autodan-automatic-and-interpretable","slug":"autodan-automatic-and-interpretable","title":"AutoDAN: Interpretable Gradient-Based Adversarial Attacks on Large Language Models","date":"2023-10-23","arxiv_id":"2310.15140","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autodan-automatic-and-interpretable#ran","syntology_url":"https://syntology.ai/paper/2310.15140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15140"}},"official":null}},{"url":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/who-is-chatgpt-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2310.01386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01386"}},"official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat","slug":"gpt-4-is-too-smart-to-be-safe-stealthy-chat","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","date":"2023-08-12","arxiv_id":"2308.06463","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat#ran","syntology_url":"https://syntology.ai/paper/2308.06463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06463"}},"official":{"repos":["robustnlp/cipherchat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"ad46fbb85610f0b63df979fb8db50ab31984d8300de06ed7db4df13ebeaf2061","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}