{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/red-teaming/papers/ran/1","list_of":"/task/red-teaming","task":"Red Teaming","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,57],"of":57,"counts":{"archive_papers_tagged":251,"with_a_code_link":110,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":251,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/red-teaming/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/we-should-identify-and-mitigate-third-party","slug":"we-should-identify-and-mitigate-third-party","title":"We Should Identify and Mitigate Third-Party Safety Risks in MCP-Powered Agent Systems","date":"2025-06-16","arxiv_id":"2506.13666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/we-should-identify-and-mitigate-third-party#ran","syntology_url":"https://syntology.ai/paper/2506.13666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13666"}},"official":{"repos":["littlelittlenine/safemcp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/redteamcua-realistic-adversarial-testing-of","slug":"redteamcua-realistic-adversarial-testing-of","title":"RedTeamCUA: Realistic Adversarial Testing of Computer-Use Agents in Hybrid Web-OS Environments","date":"2025-05-28","arxiv_id":"2505.21936","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/redteamcua-realistic-adversarial-testing-of#ran","syntology_url":"https://syntology.ai/paper/2505.21936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21936"}},"official":{"repos":["osu-nlp-group/redteamcua"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/capability-based-scaling-laws-for-llm-red","slug":"capability-based-scaling-laws-for-llm-red","title":"Capability-Based Scaling Laws for LLM Red-Teaming","date":"2025-05-26","arxiv_id":"2505.20162","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/capability-based-scaling-laws-for-llm-red#ran","syntology_url":"https://syntology.ai/paper/2505.20162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20162"}},"official":{"repos":["kotekjedi/capability-based-scaling"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/benign-samples-matter-fine-tuning-on-outlier","slug":"benign-samples-matter-fine-tuning-on-outlier","title":"Benign Samples Matter! Fine-tuning On Outlier Benign Samples Severely Breaks Safety","date":"2025-05-11","arxiv_id":"2505.06843","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benign-samples-matter-fine-tuning-on-outlier#ran","syntology_url":"https://syntology.ai/paper/2505.06843","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.06843"}},"official":{"repos":["guanzihan/benign-samples-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sudo-rm-rf-agentic-security","slug":"sudo-rm-rf-agentic-security","title":"sudo rm -rf agentic_security","date":"2025-03-26","arxiv_id":"2503.20279","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sudo-rm-rf-agentic-security#ran","syntology_url":"https://syntology.ai/paper/2503.20279","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20279"}},"official":{"repos":["AIM-Intelligence/SUDO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-balance-with-asynchrony-decoupling","slug":"trajectory-balance-with-asynchrony-decoupling","title":"Trajectory Balance with Asynchrony: Decoupling Exploration and Learning for Fast, Scalable LLM Post-Training","date":"2025-03-24","arxiv_id":"2503.18929","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-balance-with-asynchrony-decoupling#ran","syntology_url":"https://syntology.ai/paper/2503.18929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18929"}},"official":null}},{"url":"/paper/udora-a-unified-red-teaming-framework-against","slug":"udora-a-unified-red-teaming-framework-against","title":"UDora: A Unified Red Teaming Framework against LLM Agents by Dynamically Hijacking Their Own Reasoning","date":"2025-02-28","arxiv_id":"2503.01908","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/udora-a-unified-red-teaming-framework-against#ran","syntology_url":"https://syntology.ai/paper/2503.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01908"}},"official":{"repos":["ai-secure/udora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-and-enhancing-the-1","slug":"understanding-and-enhancing-the-1","title":"Understanding and Enhancing the Transferability of Jailbreaking Attacks","date":"2025-02-05","arxiv_id":"2502.03052","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understanding-and-enhancing-the-1#ran","syntology_url":"https://syntology.ai/paper/2502.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03052"}},"official":{"repos":["tmllab/2025_ICLR_PiF"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/virus-harmful-fine-tuning-attack-for-large","slug":"virus-harmful-fine-tuning-attack-for-large","title":"Virus: Harmful Fine-tuning Attack for Large Language Models Bypassing Guardrail Moderation","date":"2025-01-29","arxiv_id":"2501.17433","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/virus-harmful-fine-tuning-attack-for-large#ran","syntology_url":"https://syntology.ai/paper/2501.17433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17433"}},"official":{"repos":["git-disl/virus"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-auditing-test-to-detect-behavioral-shift","slug":"an-auditing-test-to-detect-behavioral-shift","title":"An Auditing Test To Detect Behavioral Shift in Language Models","date":"2024-10-25","arxiv_id":"2410.19406","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-auditing-test-to-detect-behavioral-shift#ran","syntology_url":"https://syntology.ai/paper/2410.19406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19406"}},"official":{"repos":["richterleo/Auditing_Test_for_LMs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset","slug":"vlfeedback-a-large-scale-ai-feedback-dataset","title":"VLFeedback: A Large-Scale AI Feedback Dataset for Large Vision-Language Models Alignment","date":"2024-10-12","arxiv_id":"2410.09421","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.09421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09421"}},"official":null}},{"url":"/paper/refusal-trained-llms-are-easily-jailbroken-as","slug":"refusal-trained-llms-are-easily-jailbroken-as","title":"Refusal-Trained LLMs Are Easily Jailbroken As Browser Agents","date":"2024-10-11","arxiv_id":"2410.13886","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/refusal-trained-llms-are-easily-jailbroken-as#ran","syntology_url":"https://syntology.ai/paper/2410.13886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13886"}},"official":{"repos":["scaleapi/browser-art"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autodan-turbo-a-lifelong-agent-for-strategy","slug":"autodan-turbo-a-lifelong-agent-for-strategy","title":"AutoDAN-Turbo: A Lifelong Agent for Strategy Self-Exploration to Jailbreak LLMs","date":"2024-10-03","arxiv_id":"2410.05295","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/autodan-turbo-a-lifelong-agent-for-strategy#ran","syntology_url":"https://syntology.ai/paper/2410.05295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05295"}},"official":{"repos":["safolab-wisc/autodan-turbo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/holistic-automated-red-teaming-for-large","slug":"holistic-automated-red-teaming-for-large","title":"Holistic Automated Red Teaming for Large Language Models through Top-Down Test Case Generation and Multi-turn Interaction","date":"2024-09-25","arxiv_id":"2409.16783","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/holistic-automated-red-teaming-for-large#ran","syntology_url":"https://syntology.ai/paper/2409.16783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16783"}},"official":{"repos":["jc-ryan/holistic_automated_red_teaming"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-adversarial-suffix-transfer","slug":"advancing-adversarial-suffix-transfer","title":"Advancing Adversarial Suffix Transfer Learning on Aligned Large Language Models","date":"2024-08-27","arxiv_id":"2408.14866","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/advancing-adversarial-suffix-transfer#ran","syntology_url":"https://syntology.ai/paper/2408.14866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14866"}},"official":{"repos":["Waffle-Liu/DeGCG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00761","slug":"2408-00761","title":"Tamper-Resistant Safeguards for Open-Weight LLMs","date":"2024-08-01","arxiv_id":"2408.00761","repositories_listed":2,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/2408-00761#ran","syntology_url":"https://syntology.ai/paper/2408.00761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00761"}},"official":{"repos":["rishub-tamirisa/tamper-resistance"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/targeted-latent-adversarial-training-improves","slug":"targeted-latent-adversarial-training-improves","title":"Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in LLMs","date":"2024-07-22","arxiv_id":"2407.15549","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/targeted-latent-adversarial-training-improves#ran","syntology_url":"https://syntology.ai/paper/2407.15549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15549"}},"official":{"repos":["aengusl/latent-adversarial-training"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reliable-and-efficient-concept-erasure-of","slug":"reliable-and-efficient-concept-erasure-of","title":"Reliable and Efficient Concept Erasure of Text-to-Image Diffusion Models","date":"2024-07-17","arxiv_id":"2407.12383","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliable-and-efficient-concept-erasure-of#ran","syntology_url":"https://syntology.ai/paper/2407.12383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12383"}},"official":{"repos":["charlesgong12/rece"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/agentpoison-red-teaming-llm-agents-via","slug":"agentpoison-red-teaming-llm-agents-via","title":"AgentPoison: Red-teaming LLM Agents via Poisoning Memory or Knowledge Bases","date":"2024-07-17","arxiv_id":"2407.12784","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agentpoison-red-teaming-llm-agents-via#ran","syntology_url":"https://syntology.ai/paper/2407.12784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12784"}},"official":{"repos":["BillChan226/AgentPoison"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sop-unlock-the-power-of-social-facilitation","slug":"sop-unlock-the-power-of-social-facilitation","title":"SeqAR: Jailbreak LLMs with Sequential Auto-Generated Characters","date":"2024-07-02","arxiv_id":"2407.01902","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sop-unlock-the-power-of-social-facilitation#ran","syntology_url":"https://syntology.ai/paper/2407.01902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01902"}},"official":{"repos":["yang-yan-yang-yan/sop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wildteaming-at-scale-from-in-the-wild","slug":"wildteaming-at-scale-from-in-the-wild","title":"WildTeaming at Scale: From In-the-Wild Jailbreaks to (Adversarially) Safer Language Models","date":"2024-06-26","arxiv_id":"2406.18510","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wildteaming-at-scale-from-in-the-wild#ran","syntology_url":"https://syntology.ai/paper/2406.18510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18510"}},"official":{"repos":["allenai/wildteaming"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/steering-without-side-effects-improving-post","slug":"steering-without-side-effects-improving-post","title":"Steering Without Side Effects: Improving Post-Deployment Control of Language Models","date":"2024-06-21","arxiv_id":"2406.15518","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/steering-without-side-effects-improving-post#ran","syntology_url":"https://syntology.ai/paper/2406.15518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15518"}},"official":{"repos":["asacooperstickland/kl-then-steer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-safety-neurons-in-large-language","slug":"finding-safety-neurons-in-large-language","title":"Finding Safety Neurons in Large Language Models","date":"2024-06-20","arxiv_id":"2406.14144","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finding-safety-neurons-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.14144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14144"}},"official":null}},{"url":"/paper/jailbreaking-as-a-reward-misspecification","slug":"jailbreaking-as-a-reward-misspecification","title":"Jailbreaking as a Reward Misspecification Problem","date":"2024-06-20","arxiv_id":"2406.14393","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":2,"n_ran_checked":4,"n_instrument":3,"n_unverified":8,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":15,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/jailbreaking-as-a-reward-misspecification#ran","syntology_url":"https://syntology.ai/paper/2406.14393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14393"}},"official":{"repos":["zhxieml/remiss-jailbreak"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/dialogue-action-tokens-steering-language","slug":"dialogue-action-tokens-steering-language","title":"Dialogue Action Tokens: Steering Language Models in Goal-Directed Dialogue with a Multi-Turn Planner","date":"2024-06-17","arxiv_id":"2406.11978","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dialogue-action-tokens-steering-language#ran","syntology_url":"https://syntology.ai/paper/2406.11978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11978"}},"official":{"repos":["likenneth/dialogue_action_token"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/csrt-evaluation-and-analysis-of-llms-using","slug":"csrt-evaluation-and-analysis-of-llms-using","title":"Code-Switching Red-Teaming: LLM Evaluation for Safety and Multilingual Understanding","date":"2024-06-17","arxiv_id":"2406.15481","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/csrt-evaluation-and-analysis-of-llms-using#ran","syntology_url":"https://syntology.ai/paper/2406.15481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15481"}},"official":{"repos":["haneul-yoo/csrt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mllmguard-a-multi-dimensional-safety","slug":"mllmguard-a-multi-dimensional-safety","title":"MLLMGuard: A Multi-dimensional Safety Evaluation Suite for Multimodal Large Language Models","date":"2024-06-11","arxiv_id":"2406.07594","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mllmguard-a-multi-dimensional-safety#ran","syntology_url":"https://syntology.ai/paper/2406.07594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07594"}},"official":{"repos":["Carol-gutianle/MLLMGuard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jailbreak-vision-language-models-via-bi-modal","slug":"jailbreak-vision-language-models-via-bi-modal","title":"Jailbreak Vision Language Models via Bi-Modal Adversarial Prompt","date":"2024-06-06","arxiv_id":"2406.04031","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jailbreak-vision-language-models-via-bi-modal#ran","syntology_url":"https://syntology.ai/paper/2406.04031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04031"}},"official":{"repos":["NY1024/BAP-Jailbreak-Vision-Language-Models-via-Bi-Modal-Adversarial-Prompt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-techniques-for-optimization-based","slug":"improved-techniques-for-optimization-based","title":"Improved Techniques for Optimization-Based Jailbreaking on Large Language Models","date":"2024-05-31","arxiv_id":"2405.21018","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/improved-techniques-for-optimization-based#ran","syntology_url":"https://syntology.ai/paper/2405.21018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.21018"}},"official":{"repos":["jiaxiaojunqaq/i-gcg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-diverse-attacks-on-large-language","slug":"learning-diverse-attacks-on-large-language","title":"Learning diverse attacks on large language models for robust red-teaming and safety tuning","date":"2024-05-28","arxiv_id":"2405.18540","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-attacks-on-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.18540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18540"}},"official":{"repos":["GFNOrg/red-teaming"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/art-automatic-red-teaming-for-text-to-image","slug":"art-automatic-red-teaming-for-text-to-image","title":"ART: Automatic Red-teaming for Text-to-Image Models to Protect Benign Users","date":"2024-05-24","arxiv_id":"2405.19360","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/art-automatic-red-teaming-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2405.19360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19360"}},"official":{"repos":["guanlinlee/art"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-inference-in-language-models","slug":"probabilistic-inference-in-language-models","title":"Probabilistic Inference in Language Models via Twisted Sequential Monte Carlo","date":"2024-04-26","arxiv_id":"2404.17546","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/probabilistic-inference-in-language-models#ran","syntology_url":"https://syntology.ai/paper/2404.17546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17546"}},"official":{"repos":["silent-zebra/twisted-smc-lm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/bias-patterns-in-the-application-of-llms-for","slug":"bias-patterns-in-the-application-of-llms-for","title":"Bias patterns in the application of LLMs for clinical decision support: A comprehensive study","date":"2024-04-23","arxiv_id":"2404.15149","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bias-patterns-in-the-application-of-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15149"}},"official":{"repos":["healthylaife/faircdsllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/advprompter-fast-adaptive-adversarial","slug":"advprompter-fast-adaptive-adversarial","title":"AdvPrompter: Fast Adaptive Adversarial Prompting for LLMs","date":"2024-04-21","arxiv_id":"2404.16873","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/advprompter-fast-adaptive-adversarial#ran","syntology_url":"https://syntology.ai/paper/2404.16873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16873"}},"official":{"repos":["facebookresearch/advprompter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/red-teaming-gpt-4v-are-gpt-4v-safe-against","slug":"red-teaming-gpt-4v-are-gpt-4v-safe-against","title":"Red Teaming GPT-4V: Are GPT-4V Safe Against Uni/Multi-Modal Jailbreak Attacks?","date":"2024-04-04","arxiv_id":"2404.03411","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/red-teaming-gpt-4v-are-gpt-4v-safe-against#ran","syntology_url":"https://syntology.ai/paper/2404.03411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03411"}},"official":{"repos":["chenxshuo/redteaminggpt4v"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tastle-distract-large-language-models-for","slug":"tastle-distract-large-language-models-for","title":"Distract Large Language Models for Automatic Jailbreak Attack","date":"2024-03-13","arxiv_id":"2403.08424","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tastle-distract-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2403.08424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08424"}},"official":{"repos":["sufenlp/AttanttionShiftJailbreak"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/red-teaming-models-for-hyperspectral-image","slug":"red-teaming-models-for-hyperspectral-image","title":"Red Teaming Models for Hyperspectral Image Analysis Using Explainable AI","date":"2024-03-12","arxiv_id":"2403.08017","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/red-teaming-models-for-hyperspectral-image#ran","syntology_url":"https://syntology.ai/paper/2403.08017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08017"}},"official":null}},{"url":"/paper/aligners-decoupling-llms-and-alignment","slug":"aligners-decoupling-llms-and-alignment","title":"Aligners: Decoupling LLMs and Alignment","date":"2024-03-07","arxiv_id":"2403.04224","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligners-decoupling-llms-and-alignment#ran","syntology_url":"https://syntology.ai/paper/2403.04224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04224"}},"official":{"repos":["lilianngweta/aligners-and-inspectors","lilianngweta/aligners"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curiosity-driven-red-teaming-for-large","slug":"curiosity-driven-red-teaming-for-large","title":"Curiosity-driven Red-teaming for Large Language Models","date":"2024-02-29","arxiv_id":"2402.19464","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curiosity-driven-red-teaming-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.19464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19464"}},"official":{"repos":["improbable-ai/curiosity_redteam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-smith-a-single-image-can-jailbreak-one","slug":"agent-smith-a-single-image-can-jailbreak-one","title":"Agent Smith: A Single Image Can Jailbreak One Million Multimodal LLM Agents Exponentially Fast","date":"2024-02-13","arxiv_id":"2402.08567","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agent-smith-a-single-image-can-jailbreak-one#ran","syntology_url":"https://syntology.ai/paper/2402.08567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08567"}},"official":{"repos":["sail-sg/agent-smith"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/harmbench-a-standardized-evaluation-framework","slug":"harmbench-a-standardized-evaluation-framework","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","date":"2024-02-06","arxiv_id":"2402.04249","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harmbench-a-standardized-evaluation-framework#ran","syntology_url":"https://syntology.ai/paper/2402.04249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04249"}},"official":{"repos":["centerforaisafety/harmbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causality-analysis-for-evaluating-the","slug":"causality-analysis-for-evaluating-the","title":"Causality Analysis for Evaluating the Security of Large Language Models","date":"2023-12-13","arxiv_id":"2312.07876","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/causality-analysis-for-evaluating-the#ran","syntology_url":"https://syntology.ai/paper/2312.07876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07876"}},"official":{"repos":["casperllm/casper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-control-improving-safety-despite","slug":"ai-control-improving-safety-despite","title":"AI Control: Improving Safety Despite Intentional Subversion","date":"2023-12-12","arxiv_id":"2312.06942","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai-control-improving-safety-despite#ran","syntology_url":"https://syntology.ai/paper/2312.06942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06942"}},"official":{"repos":["rgreenblatt/control-evaluations"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backdoor-activation-attack-attack-large","slug":"backdoor-activation-attack-attack-large","title":"Trojan Activation Attack: Red-Teaming Large Language Models using Activation Steering for Safety-Alignment","date":"2023-11-15","arxiv_id":"2311.09433","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/backdoor-activation-attack-attack-large#ran","syntology_url":"https://syntology.ai/paper/2311.09433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09433"}},"official":{"repos":["wang2226/backdoor-activation-attack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stealthy-and-persistent-unalignment-on-large","slug":"stealthy-and-persistent-unalignment-on-large","title":"Stealthy and Persistent Unalignment on Large Language Models via Backdoor Injections","date":"2023-11-15","arxiv_id":"2312.00027","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stealthy-and-persistent-unalignment-on-large#ran","syntology_url":"https://syntology.ai/paper/2312.00027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00027"}},"official":{"repos":["caoyuanpu/backdoorunalign"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/attack-prompt-generation-for-red-teaming-and","slug":"attack-prompt-generation-for-red-teaming-and","title":"Attack Prompt Generation for Red Teaming and Defending Large Language Models","date":"2023-10-19","arxiv_id":"2310.12505","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attack-prompt-generation-for-red-teaming-and#ran","syntology_url":"https://syntology.ai/paper/2310.12505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12505"}},"official":{"repos":["aatrox103/sap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ring-a-bell-how-reliable-are-concept-removal","slug":"ring-a-bell-how-reliable-are-concept-removal","title":"Ring-A-Bell! How Reliable are Concept Removal Methods for Diffusion Models?","date":"2023-10-16","arxiv_id":"2310.10012","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/ring-a-bell-how-reliable-are-concept-removal#ran","syntology_url":"https://syntology.ai/paper/2310.10012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10012"}},"official":{"repos":["chiayi-hsu/ring-a-bell"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/catastrophic-jailbreak-of-open-source-llms","slug":"catastrophic-jailbreak-of-open-source-llms","title":"Catastrophic Jailbreak of Open-source LLMs via Exploiting Generation","date":"2023-10-10","arxiv_id":"2310.06987","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/catastrophic-jailbreak-of-open-source-llms#ran","syntology_url":"https://syntology.ai/paper/2310.06987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06987"}},"official":{"repos":["princeton-sysml/jailbreak_llm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gptfuzzer-red-teaming-large-language-models","slug":"gptfuzzer-red-teaming-large-language-models","title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","date":"2023-09-19","arxiv_id":"2309.10253","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gptfuzzer-red-teaming-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2309.10253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10253"}},"official":{"repos":["sherdencooper/gptfuzz"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting4debugging-red-teaming-text-to-image","slug":"prompting4debugging-red-teaming-text-to-image","title":"Prompting4Debugging: Red-Teaming Text-to-Image Diffusion Models by Finding Problematic Prompts","date":"2023-09-12","arxiv_id":"2309.06135","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompting4debugging-red-teaming-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2309.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06135"}},"official":{"repos":["joycenerd/p4d"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat","slug":"gpt-4-is-too-smart-to-be-safe-stealthy-chat","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","date":"2023-08-12","arxiv_id":"2308.06463","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat#ran","syntology_url":"https://syntology.ai/paper/2308.06463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06463"}},"official":{"repos":["robustnlp/cipherchat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xstest-a-test-suite-for-identifying","slug":"xstest-a-test-suite-for-identifying","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","date":"2023-08-02","arxiv_id":"2308.01263","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/xstest-a-test-suite-for-identifying#ran","syntology_url":"https://syntology.ai/paper/2308.01263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01263"}},"official":{"repos":["paul-rottger/exaggerated-safety"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/red-teaming-language-model-detectors-with","slug":"red-teaming-language-model-detectors-with","title":"Red Teaming Language Model Detectors with Language Models","date":"2023-05-31","arxiv_id":"2305.19713","repositories_listed":2,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/red-teaming-language-model-detectors-with#ran","syntology_url":"https://syntology.ai/paper/2305.19713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19713"}},"official":{"repos":["shizhouxing/Attack-LM-Detectors"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/query-efficient-black-box-red-teaming-via","slug":"query-efficient-black-box-red-teaming-via","title":"Query-Efficient Black-Box Red Teaming via Bayesian Optimization","date":"2023-05-27","arxiv_id":"2305.17444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-efficient-black-box-red-teaming-via#ran","syntology_url":"https://syntology.ai/paper/2305.17444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17444"}},"official":{"repos":["snu-mllab/bayesian-red-teaming"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/white-box-adversarial-policies-in-deep","slug":"white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/white-box-adversarial-policies-in-deep#ran","syntology_url":"https://syntology.ai/paper/2209.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02167"}},"official":{"repos":["thestephencasper/lm_white_box_attacks","thestephencasper/white_box_rarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/red-teaming-language-models-to-reduce-harms","slug":"red-teaming-language-models-to-reduce-harms","title":"Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","date":"2022-08-23","arxiv_id":"2209.07858","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/red-teaming-language-models-to-reduce-harms#ran","syntology_url":"https://syntology.ai/paper/2209.07858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07858"}},"official":{"repos":["anthropics/hh-rlhf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/red-teaming-language-models-with-language","slug":"red-teaming-language-models-with-language","title":"Red Teaming Language Models with Language Models","date":"2022-02-07","arxiv_id":"2202.03286","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/red-teaming-language-models-with-language#ran","syntology_url":"https://syntology.ai/paper/2202.03286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03286"}},"official":null}}],"record_sha256":"b305e407969edc5d9ea714c60bd01de34ac1e58c9109596212ab478f4d8e1aa5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}