{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-score","entry":"extract_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":18,"n_papers_ran":12,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":20,"n_samples_ran":14,"n_samples_fingerprinted":9,"n_places":20,"n_places_pointer_only":8,"by_status":{"ran_honours":4,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":10,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.25152","paper":"/paper/arxiv-2606-25152","title":"Hitting a Moving Target: Test-Time Adaptation for AI Text Detection under Continual Distribution Shift","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"kkr36/llm_detection","path":"arxiv/inference_set_rewrite/iterative_prompt_rewrite_scale/llm_judge.py","file_url":"https://github.com/kkr36/llm_detection/blob/HEAD/arxiv/inference_set_rewrite/iterative_prompt_rewrite_scale/llm_judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2706b98edd8c58b","mcp_get_code":{"code_sha256":"a2706b98edd8c58b"}},{"arxiv_id":"2606.07970","paper":"/paper/arxiv-2606-07970","title":"Defending Against Malicious Finetuning by Scaling Train-time Adversarial Attacks","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"haomingwen/patcher","path":"evaluate/gpt_evaluate.py","file_url":"https://github.com/haomingwen/patcher/blob/HEAD/evaluate/gpt_evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1e9b2b7a8f925a69","mcp_get_code":{"code_sha256":"1e9b2b7a8f925a69"}},{"arxiv_id":"2604.27453","paper":"/paper/arxiv-2604-27453","title":"From Coarse to Fine: Benchmarking and Reward Modeling for Writing-Centric Generation Tasks","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rainier-rq1/From_Coarse_to_Fine","path":"WEval/WritingBench-Critic.py","file_url":"https://github.com/Rainier-rq1/From_Coarse_to_Fine/blob/HEAD/WEval/WritingBench-Critic.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7da4354e2c6f6c2c","mcp_get_code":{"code_sha256":"7da4354e2c6f6c2c"}},{"arxiv_id":"2604.27453","paper":"/paper/arxiv-2604-27453","title":"From Coarse to Fine: Benchmarking and Reward Modeling for Writing-Centric Generation Tasks","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rainier-rq1/From_Coarse_to_Fine","path":"WEval/llm_judge.py","file_url":"https://github.com/Rainier-rq1/From_Coarse_to_Fine/blob/HEAD/WEval/llm_judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2fd3492b9408427","mcp_get_code":{"code_sha256":"a2fd3492b9408427"}},{"arxiv_id":"2602.03548","paper":"/paper/arxiv-2602-03548","title":"SEAD: Self-Evolving Agent for Multi-Turn Service Dialogue","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Da1yuqin/SEAD","path":"utils/analyze_chatbot_mistakes.py","file_url":"https://github.com/Da1yuqin/SEAD/blob/HEAD/utils/analyze_chatbot_mistakes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb8290dd4e4e685b","mcp_get_code":{"code_sha256":"fb8290dd4e4e685b"}},{"arxiv_id":"2601.09270","paper":"/paper/arxiv-2601-09270","title":"MCGA: A Multi-task Classical Chinese Literary Genre Audio Corpus","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"yxduir/MCGA","path":"eval/eval_model.py","file_url":"https://github.com/yxduir/MCGA/blob/HEAD/eval/eval_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b5f599588a4adc4a","mcp_get_code":{"code_sha256":"b5f599588a4adc4a"}},{"arxiv_id":"2510.20956","paper":"/paper/arxiv-2510-20956","title":"Self-Jailbreaking: Language Models Can Reason Themselves Out of Safety Alignment After Benign Reasoning Training","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"BatsResearch/self-jailbreaking","path":"scripts/eval/strongreject_eval.py","file_url":"https://github.com/BatsResearch/self-jailbreaking/blob/HEAD/scripts/eval/strongreject_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e3b7df9a5aa407fe","mcp_get_code":{"code_sha256":"e3b7df9a5aa407fe"}},{"arxiv_id":"2505.17018","paper":"/paper/sophiavl-r1-reinforcing-mllms-reasoning-with","title":"SophiaVL-R1: Reinforcing MLLMs Reasoning with Thinking Reward","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kxfan2002/sophiavl-r1","path":"verl/workers/reward/custom.py","file_url":"https://github.com/kxfan2002/sophiavl-r1/blob/HEAD/verl/workers/reward/custom.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"41ec901c077b9860","mcp_get_code":{"code_sha256":"41ec901c077b9860"}},{"arxiv_id":"2505.16211","paper":"/paper/audiotrust-benchmarking-the-multifaceted","title":"AudioTrust: Benchmarking the Multifaceted Trustworthiness of Audio Large Language Models","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jusperlee/audiotrust","path":"audio_evals/eval_task.py","file_url":"https://github.com/jusperlee/audiotrust/blob/HEAD/audio_evals/eval_task.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"67daa768dfd59db2","mcp_get_code":{"code_sha256":"67daa768dfd59db2"}},{"arxiv_id":"2505.02370","paper":"/paper/superedit-rectifying-and-facilitating","title":"SuperEdit: Rectifying and Facilitating Supervision for Instruction-Based Image Editing","date":"2025-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/superedit","path":"eval/eval_instructpix2pix.py","file_url":"https://github.com/bytedance/superedit/blob/HEAD/eval/eval_instructpix2pix.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ea7b78776c588d7f","mcp_get_code":{"code_sha256":"ea7b78776c588d7f"}},{"arxiv_id":"2503.08741","paper":"/paper/oasis-one-image-is-all-you-need-for","title":"Oasis: One Image is All You Need for Multimodal Instruction Data Synthesis","date":"2025-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Letian2003/MM_INF","path":"pp_func/oasis.py","file_url":"https://github.com/Letian2003/MM_INF/blob/HEAD/pp_func/oasis.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c39c295401d2dd49","mcp_get_code":{"code_sha256":"c39c295401d2dd49"}},{"arxiv_id":"2502.16182","paper":"/paper/ipo-your-language-model-is-secretly-a","title":"IPO: Your Language Model is Secretly a Preference Classifier","date":"2025-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shivank21/Implicit_Preference_Optimization","path":"Preference_Comparison/RM_Bench/self_reward.py","file_url":"https://github.com/shivank21/Implicit_Preference_Optimization/blob/HEAD/Preference_Comparison/RM_Bench/self_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7204c024b35e7517","mcp_get_code":{"code_sha256":"7204c024b35e7517"}},{"arxiv_id":"2502.03052","paper":"/paper/understanding-and-enhancing-the-1","title":"Understanding and Enhancing the Transferability of Jailbreaking Attacks","date":"2025-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tmllab/2025_ICLR_PiF","path":"attack_clm.py","file_url":"https://github.com/tmllab/2025_ICLR_PiF/blob/HEAD/attack_clm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"21b2a816defb053b","mcp_get_code":{"code_sha256":"21b2a816defb053b"}},{"arxiv_id":"2407.19633","paper":"/paper/optimus-0-3-using-large-language-models-to","title":"OptiMUS-0.3: Using Large Language Models to Model and Solve Optimization Problems at Scale","date":"2024-07-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teshnizi/optimus","path":"parameters.py","file_url":"https://github.com/teshnizi/optimus/blob/HEAD/parameters.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"edbd5ba229bb9dfc","mcp_get_code":{"code_sha256":"edbd5ba229bb9dfc"}},{"arxiv_id":"2407.19633","paper":"/paper/optimus-0-3-using-large-language-models-to","title":"OptiMUS-0.3: Using Large Language Models to Model and Solve Optimization Problems at Scale","date":"2024-07-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teshnizi/optimus","path":"variables.py","file_url":"https://github.com/teshnizi/optimus/blob/HEAD/variables.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"90629c8ff26c53f1","mcp_get_code":{"code_sha256":"90629c8ff26c53f1"}},{"arxiv_id":"2407.02099","paper":"/paper/helpful-assistant-or-fruitful-facilitator","title":"Helpful assistant or fruitful facilitator? Investigating how personas affect language model behavior","date":"2024-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"peluz/persona-behavior","path":"compute_results.py","file_url":"https://github.com/peluz/persona-behavior/blob/HEAD/compute_results.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0dbfb54acfddc8c8","mcp_get_code":{"code_sha256":"0dbfb54acfddc8c8"}},{"arxiv_id":"2406.18522","paper":"/paper/chronomagic-bench-a-benchmark-for-metamorphic","title":"ChronoMagic-Bench: A Benchmark for Metamorphic Evaluation of Text-to-Time-lapse Video Generation","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PKU-YuanGroup/ChronoMagic-Bench","path":"UMT/step3_get_merge_umt_scores.py","file_url":"https://github.com/PKU-YuanGroup/ChronoMagic-Bench/blob/HEAD/UMT/step3_get_merge_umt_scores.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"690f98d4b6d870df","mcp_get_code":{"code_sha256":"690f98d4b6d870df"}},{"arxiv_id":"2405.19744","paper":"/paper/x-instruction-aligning-language-model-in-low","title":"X-Instruction: Aligning Language Model in Low-resource Languages with Self-curated Cross-lingual Instructions","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"znlp/x-instruction","path":"utils/gpt4_eval.py","file_url":"https://github.com/znlp/x-instruction/blob/HEAD/utils/gpt4_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c70b233c8023d8d3","mcp_get_code":{"code_sha256":"c70b233c8023d8d3"}},{"arxiv_id":"2401.10019","paper":"/paper/r-judge-benchmarking-safety-risk-awareness","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","date":"2024-01-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lordog/r-judge","path":"eval/risk_identification.py","file_url":"https://github.com/lordog/r-judge/blob/HEAD/eval/risk_identification.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"38ff8148457f4eff","mcp_get_code":{"code_sha256":"38ff8148457f4eff"}},{"arxiv_id":"openreview_JjILY9i6Wi","paper":null,"title":"arXiv:openreview_JjILY9i6Wi","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"nwlt/IPMark","path":"baseline/baseline_attack_detect.py","file_url":"https://github.com/nwlt/IPMark/blob/HEAD/baseline/baseline_attack_detect.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ce1677e395105a2","mcp_get_code":{"code_sha256":"6ce1677e395105a2"}}]}