{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-prompts","entry":"load_prompts","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":41,"n_papers_ran":24,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":37,"n_samples_ran":22,"n_samples_fingerprinted":0,"n_places":42,"n_places_pointer_only":19,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":8,"ran_fixture":0,"ran":14,"unverified":15},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.30415","paper":"/paper/arxiv-2605-30415","title":"Domain Adaptation and Reasoning Frameworks in Language Models: A Controlled Experiment with Historical Cosmology","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"fdeberna/chat-ptolemaic","path":"evaluation/generate_eval.py","file_url":"https://github.com/fdeberna/chat-ptolemaic/blob/HEAD/evaluation/generate_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4027d245a8e4b2a5","mcp_get_code":{"code_sha256":"4027d245a8e4b2a5"}},{"arxiv_id":"2603.03081","paper":"/paper/arxiv-2603-03081","title":"TAO-Attack: Toward Advanced Optimization-Based Jailbreak Attacks for Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"ZevineXu/TAO-Attack","path":"api_experiments/evaluate_api_models.py","file_url":"https://github.com/ZevineXu/TAO-Attack/blob/HEAD/api_experiments/evaluate_api_models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"645890ba00019a69","mcp_get_code":{"code_sha256":"645890ba00019a69"}},{"arxiv_id":"2602.07422","paper":"/paper/arxiv-2602-07422","title":"Secure Code Generation via Online Reinforcement Learning with Vulnerability Reward Model","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"AndrewWTY/SecCoderX","path":"vul_induce_prompt_pipeline/generate_instructions_batch.py","file_url":"https://github.com/AndrewWTY/SecCoderX/blob/HEAD/vul_induce_prompt_pipeline/generate_instructions_batch.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"075ea2984064fe95","mcp_get_code":{"code_sha256":"075ea2984064fe95"}},{"arxiv_id":"2601.17230","paper":"/paper/arxiv-2601-17230","title":"CaseFacts: A Benchmark for Legal Fact-Checking and Precedent Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"idirlab/CaseFacts","path":"experiments/batch_inference_moe.py","file_url":"https://github.com/idirlab/CaseFacts/blob/HEAD/experiments/batch_inference_moe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"824e0abf76e43ed8","mcp_get_code":{"code_sha256":"824e0abf76e43ed8"}},{"arxiv_id":"2601.09270","paper":"/paper/arxiv-2601-09270","title":"MCGA: A Multi-task Classical Chinese Literary Genre Audio Corpus","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"yxduir/MCGA","path":"eval/api_mcga.py","file_url":"https://github.com/yxduir/MCGA/blob/HEAD/eval/api_mcga.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f5df9a3ab0542ab1","mcp_get_code":{"code_sha256":"f5df9a3ab0542ab1"}},{"arxiv_id":"2509.13760","paper":"/paper/arxiv-2509-13760","title":"Iterative Prompt Refinement for Safer Text-to-Image Generation","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"ku-dmlab/IPR","path":"vl_rl.py","file_url":"https://github.com/ku-dmlab/IPR/blob/HEAD/vl_rl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e9c9a97fe7d56a76","mcp_get_code":{"code_sha256":"e9c9a97fe7d56a76"}},{"arxiv_id":"2505.20275","paper":"/paper/imgedit-a-unified-image-editing-dataset-and","title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pku-yuangroup/imgedit","path":"Benchmark/Basic/basic_bench.py","file_url":"https://github.com/pku-yuangroup/imgedit/blob/HEAD/Benchmark/Basic/basic_bench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"913a857e957a8c91","mcp_get_code":{"code_sha256":"913a857e957a8c91"}},{"arxiv_id":"2505.16239","paper":"/paper/dove-efficient-one-step-diffusion-model-for","title":"DOVE: Efficient One-Step Diffusion Model for Real-World Video Super-Resolution","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhengchen1999/DOVE","path":"finetune/datasets/utils.py","file_url":"https://github.com/zhengchen1999/DOVE/blob/HEAD/finetune/datasets/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fad222ebf1926ea1","mcp_get_code":{"code_sha256":"fad222ebf1926ea1"}},{"arxiv_id":"2504.10637","paper":"/paper/better-estimation-of-the-kl-divergence","title":"Better Estimation of the KL Divergence Between Language Models","date":"2025-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rycolab/kl-rb","path":"rloo_sentiment.py","file_url":"https://github.com/rycolab/kl-rb/blob/HEAD/rloo_sentiment.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c77fc1b61189419","mcp_get_code":{"code_sha256":"8c77fc1b61189419"}},{"arxiv_id":"2503.09601","paper":"/paper/rewardsds-aligning-score-distillation-via","title":"RewardSDS: Aligning Score Distillation via Reward-Weighted Sampling","date":"2025-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"itaychachy/RewardSDS","path":"evaluation/aesthetic_eval.py","file_url":"https://github.com/itaychachy/RewardSDS/blob/HEAD/evaluation/aesthetic_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d98daaa9e9aeb095","mcp_get_code":{"code_sha256":"d98daaa9e9aeb095"}},{"arxiv_id":"2503.00948","paper":"/paper/extrapolating-and-decoupling-image-to-video","title":"Extrapolating and Decoupling Image-to-Video Generation Models: Motion Modeling is Easier Than You Think","date":"2025-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Chuge0335/EDG","path":"evaluation/infer_multi.py","file_url":"https://github.com/Chuge0335/EDG/blob/HEAD/evaluation/infer_multi.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea66ba28a2e932e1","mcp_get_code":{"code_sha256":"ea66ba28a2e932e1"}},{"arxiv_id":"2502.06352","paper":"/paper/lantern-enhanced-relaxed-speculative-decoding","title":"LANTERN++: Enhancing Relaxed Speculative Decoding with Static Tree Drafting for Visual Auto-regressive Models","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jadohu/lantern","path":"entrypoints/generate_images.py","file_url":"https://github.com/jadohu/lantern/blob/HEAD/entrypoints/generate_images.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"865916f0c52143f1","mcp_get_code":{"code_sha256":"865916f0c52143f1"}},{"arxiv_id":"2412.19645","paper":"/paper/videomaker-zero-shot-customized-video","title":"VideoMaker: Zero-shot Customized Video Generation with the Inherent Force of Video Diffusion Models","date":"2024-12-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wutao-cs/videomaker","path":"inference.py","file_url":"https://github.com/wutao-cs/videomaker/blob/HEAD/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ea66ba28a2e932e1","mcp_get_code":{"code_sha256":"ea66ba28a2e932e1"}},{"arxiv_id":"2410.19109","paper":"/paper/rsa-control-a-pragmatics-grounded-lightweight","title":"RSA-Control: A Pragmatics-Grounded Lightweight Controllable Text Generation Framework","date":"2024-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ewanwong/rsa-control","path":"toxicity_bias_mitigation/io_utils.py","file_url":"https://github.com/ewanwong/rsa-control/blob/HEAD/toxicity_bias_mitigation/io_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"44bbf4459b4f2e94","mcp_get_code":{"code_sha256":"44bbf4459b4f2e94"}},{"arxiv_id":"2409.15380","paper":"/paper/kalahi-a-handcrafted-grassroots-cultural-llm","title":"Kalahi: A handcrafted, grassroots cultural LLM evaluation suite for Filipino","date":"2024-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aisingapore/kalahi","path":"kalahi/utilities.py","file_url":"https://github.com/aisingapore/kalahi/blob/HEAD/kalahi/utilities.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"a018415e35b1d137","mcp_get_code":{"code_sha256":"a018415e35b1d137"}},{"arxiv_id":"2409.11219","paper":"/paper/score-forgetting-distillation-a-swift-data","title":"Score Forgetting Distillation: A Swift, Data-Free Method for Machine Unlearning in Diffusion Models","date":"2024-09-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-research/i2p","path":"eval/q16.py","file_url":"https://github.com/ml-research/i2p/blob/HEAD/eval/q16.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19896f44b0b6ff3e","mcp_get_code":{"code_sha256":"19896f44b0b6ff3e"}},{"arxiv_id":"2408.08661","paper":"/paper/mia-tuner-adapting-large-language-models-as","title":"MIA-Tuner: Adapting Large Language Models as Pre-training Text Detector","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wjfu99/mia-tuner","path":"my_utils.py","file_url":"https://github.com/wjfu99/mia-tuner/blob/HEAD/my_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"91d1d9b0c9c266d7","mcp_get_code":{"code_sha256":"91d1d9b0c9c266d7"}},{"arxiv_id":"2408.06072","paper":"/paper/cogvideox-text-to-video-diffusion-models-with","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","date":"2024-08-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/cogvideo","path":"finetune/datasets/utils.py","file_url":"https://github.com/thudm/cogvideo/blob/HEAD/finetune/datasets/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fad222ebf1926ea1","mcp_get_code":{"code_sha256":"fad222ebf1926ea1"}},{"arxiv_id":"2407.15240","paper":"/paper/bigbench-a-unified-benchmark-for-social-bias","title":"BIGbench: A Unified Benchmark for Evaluating Multi-dimensional Social Biases in Text-to-Image Models","date":"2024-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigbench2024/bigbench2024","path":"benchmark/generate/generate.py","file_url":"https://github.com/bigbench2024/bigbench2024/blob/HEAD/benchmark/generate/generate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"a2205613ccd243e1","mcp_get_code":{"code_sha256":"a2205613ccd243e1"}},{"arxiv_id":"2407.09121","paper":"/paper/refuse-whenever-you-feel-unsafe-improving","title":"Refuse Whenever You Feel Unsafe: Improving Safety in LLMs via Decoupled Refusal Training","date":"2024-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"robustnlp/derta","path":"evaluation.py","file_url":"https://github.com/robustnlp/derta/blob/HEAD/evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ce51d685abf1ad76","mcp_get_code":{"code_sha256":"ce51d685abf1ad76"}},{"arxiv_id":"2406.01288","paper":"/paper/improved-few-shot-jailbreaking-can-circumvent","title":"Improved Few-Shot Jailbreaking Can Circumvent Aligned Language Models and Their Defenses","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/I-FSJ","path":"api_experiments/evaluate_api_models.py","file_url":"https://github.com/sail-sg/I-FSJ/blob/HEAD/api_experiments/evaluate_api_models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"645890ba00019a69","mcp_get_code":{"code_sha256":"645890ba00019a69"}},{"arxiv_id":"2406.00380","paper":"/paper/the-best-of-both-worlds-toward-an-honest-and","title":"HonestLLM: Toward an Honest and Helpful Large Language Model","date":"2024-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Flossiee/HonestyLLM","path":"training_free/prompt_loader.py","file_url":"https://github.com/Flossiee/HonestyLLM/blob/HEAD/training_free/prompt_loader.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c80548e335885a3e","mcp_get_code":{"code_sha256":"c80548e335885a3e"}},{"arxiv_id":"2405.17814","paper":"/paper/faintbench-a-holistic-and-precise-benchmark","title":"FAIntbench: A Holistic and Precise Benchmark for Bias Evaluation in Text-to-Image Models","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"astarojth/faintbench-v1","path":"comfyui/generate_lcm.py","file_url":"https://github.com/astarojth/faintbench-v1/blob/HEAD/comfyui/generate_lcm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"632c123d9b1dc4f8","mcp_get_code":{"code_sha256":"632c123d9b1dc4f8"}},{"arxiv_id":"2405.10548","paper":"/paper/language-models-can-exploit-cross-task-in","title":"Language Models can Exploit Cross-Task In-context Learning for Data-Scarce Novel Tasks","date":"2024-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"C-anwoy/Cross-Task-ICL","path":"utils/data.py","file_url":"https://github.com/C-anwoy/Cross-Task-ICL/blob/HEAD/utils/data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e401d40128e5c42a","mcp_get_code":{"code_sha256":"e401d40128e5c42a"}},{"arxiv_id":"2405.10311","paper":"/paper/unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"castorini/unirag","path":"src/unirag/eval_image_generation.py","file_url":"https://github.com/castorini/unirag/blob/HEAD/src/unirag/eval_image_generation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1d3391d39efd8b4c","mcp_get_code":{"code_sha256":"1d3391d39efd8b4c"}},{"arxiv_id":"2405.03000","paper":"/paper/medadapter-efficient-test-time-adaptation-of","title":"MedAdapter: Efficient Test-Time Adaptation of Large Language Models towards Medical Reasoning","date":"2024-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wshi83/MedAdapter","path":"reward_model/orm/orm_guide.py","file_url":"https://github.com/wshi83/MedAdapter/blob/HEAD/reward_model/orm/orm_guide.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"537590c07503fa34","mcp_get_code":{"code_sha256":"537590c07503fa34"}},{"arxiv_id":"2404.13968","paper":"/paper/protecting-your-llms-with-information","title":"Protecting Your LLMs with Information Bottleneck","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-attacks/llm-attacks","path":"api_experiments/evaluate_api_models.py","file_url":"https://github.com/llm-attacks/llm-attacks/blob/HEAD/api_experiments/evaluate_api_models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"645890ba00019a69","mcp_get_code":{"code_sha256":"645890ba00019a69"}},{"arxiv_id":"2403.11322","paper":"/paper/stateflow-enhancing-llm-task-solving-through","title":"StateFlow: Enhancing LLM Task-Solving through State-Driven Workflows","date":"2024-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevin666aa/stateflow","path":"ALFWorld/src/chat_utils.py","file_url":"https://github.com/kevin666aa/stateflow/blob/HEAD/ALFWorld/src/chat_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ad7277e0c33fc9aa","mcp_get_code":{"code_sha256":"ad7277e0c33fc9aa"}},{"arxiv_id":"2402.16459","paper":"/paper/defending-llms-against-jailbreaking-attacks","title":"Defending LLMs against Jailbreaking Attacks via Backtranslation","date":"2024-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yihanwang617/llm-jailbreaking-defense-backtranslation","path":"utils.py","file_url":"https://github.com/yihanwang617/llm-jailbreaking-defense-backtranslation/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"37cddda277e8045d","mcp_get_code":{"code_sha256":"37cddda277e8045d"}},{"arxiv_id":"2402.12991","paper":"/paper/trap-targeted-random-adversarial-prompt","title":"TRAP: Targeted Random Adversarial Prompt Honeypot for Black-Box Identification","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"framartin/trap","path":"detect_llm/baseline_ppl.py","file_url":"https://github.com/framartin/trap/blob/HEAD/detect_llm/baseline_ppl.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7dc4896ac741214","mcp_get_code":{"code_sha256":"c7dc4896ac741214"}},{"arxiv_id":"2402.12991","paper":"/paper/trap-targeted-random-adversarial-prompt","title":"TRAP: Targeted Random Adversarial Prompt Honeypot for Black-Box Identification","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"parameterlab/trap","path":"llm_attacks/api_experiments/evaluate_api_models.py","file_url":"https://github.com/parameterlab/trap/blob/HEAD/llm_attacks/api_experiments/evaluate_api_models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"645890ba00019a69","mcp_get_code":{"code_sha256":"645890ba00019a69"}},{"arxiv_id":"2311.09718","paper":"/paper/you-don-t-need-a-personality-test-to-know","title":"You don't need a personality test to know these models are unreliable: Assessing the Reliability of Large Language Models on Psychometric Instruments","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"orange0629/llm-personas","path":"src/prompt-variation/run_prompts_on_models.py","file_url":"https://github.com/orange0629/llm-personas/blob/HEAD/src/prompt-variation/run_prompts_on_models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"872da3bc878987a3","mcp_get_code":{"code_sha256":"872da3bc878987a3"}},{"arxiv_id":"2310.15773","paper":"/paper/bless-benchmarking-large-language-models-on","title":"BLESS: Benchmarking Large Language Models on Sentence Simplification","date":"2023-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZurichNLP/BLESS","path":"utils/helpers.py","file_url":"https://github.com/ZurichNLP/BLESS/blob/HEAD/utils/helpers.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2360f86ab2cb2b0","mcp_get_code":{"code_sha256":"d2360f86ab2cb2b0"}},{"arxiv_id":"2308.00113","paper":"/paper/three-bricks-to-consolidate-watermarks-for","title":"Three Bricks to Consolidate Watermarks for Large Language Models","date":"2023-07-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/three_bricks","path":"main_watermark.py","file_url":"https://github.com/facebookresearch/three_bricks/blob/HEAD/main_watermark.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e43d8e6c09e3a48a","mcp_get_code":{"code_sha256":"e43d8e6c09e3a48a"}},{"arxiv_id":"2307.05977","paper":"/paper/towards-safe-self-distillation-of-internet","title":"Towards Safe Self-Distillation of Internet-Scale Text-to-Image Diffusion Models","date":"2023-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nannullna/safe-diffusion","path":"generate.py","file_url":"https://github.com/nannullna/safe-diffusion/blob/HEAD/generate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f927d43843c43af","mcp_get_code":{"code_sha256":"9f927d43843c43af"}},{"arxiv_id":"2305.04175","paper":"/paper/text-to-image-diffusion-models-can-be-easily","title":"Text-to-Image Diffusion Models can be Easily Backdoored through Multimodal Data Poisoning","date":"2023-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sf-zhai/badt2i","path":"evaluation/img_gen.py","file_url":"https://github.com/sf-zhai/badt2i/blob/HEAD/evaluation/img_gen.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bbd2ac7ff643d1cf","mcp_get_code":{"code_sha256":"bbd2ac7ff643d1cf"}},{"arxiv_id":"2212.10264","paper":"/paper/recode-robustness-evaluation-of-code","title":"ReCode: Robustness Evaluation of Code Generation Models","date":"2022-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"frabbisw/robustextended","path":"evalplus/evaluate_inputs.py","file_url":"https://github.com/frabbisw/robustextended/blob/HEAD/evalplus/evaluate_inputs.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d459632bb421c25b","mcp_get_code":{"code_sha256":"d459632bb421c25b"}},{"arxiv_id":"2205.11601","paper":"/paper/challenges-in-measuring-bias-via-open-ended","title":"Challenges in Measuring Bias via Open-Ended Language Generation","date":"2022-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"feyzaakyurek/bias-textgen","path":"complete_prompts.py","file_url":"https://github.com/feyzaakyurek/bias-textgen/blob/HEAD/complete_prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"75208746d14a8b86","mcp_get_code":{"code_sha256":"75208746d14a8b86"}},{"arxiv_id":"2202.07646","paper":"/paper/quantifying-memorization-across-neural","title":"Quantifying Memorization Across Neural Language Models","date":"2022-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ftramer/lm-extraction-benchmark","path":"baseline/simple_baseline.py","file_url":"https://github.com/ftramer/lm-extraction-benchmark/blob/HEAD/baseline/simple_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"53e3e08755cfa895","mcp_get_code":{"code_sha256":"53e3e08755cfa895"}},{"arxiv_id":"1911.02549","paper":"/paper/mlperf-inference-benchmark","title":"MLPerf Inference Benchmark","date":"2019-11-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlcommons/inference","path":"text_to_video/wan-2.2-t2v-a14b/run_inference.py","file_url":"https://github.com/mlcommons/inference/blob/HEAD/text_to_video/wan-2.2-t2v-a14b/run_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7b3e0f2610f104fb","mcp_get_code":{"code_sha256":"7b3e0f2610f104fb"}},{"arxiv_id":"openreview_47JZSOkw5C","paper":null,"title":"arXiv:openreview_47JZSOkw5C","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"bxuanz/SigMa","path":"run_sigma_qwen.py","file_url":"https://github.com/bxuanz/SigMa/blob/HEAD/run_sigma_qwen.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"63ccd86a50d62f8e","mcp_get_code":{"code_sha256":"63ccd86a50d62f8e"}},{"arxiv_id":"2025.naacl-long.14","paper":null,"title":"arXiv:2025.naacl-long.14","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"vectara/mirage-bench","path":"mirage_bench/util.py","file_url":"https://github.com/vectara/mirage-bench/blob/HEAD/mirage_bench/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ee845a8455f6d910","mcp_get_code":{"code_sha256":"ee845a8455f6d910"}}]}