{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/calculate-score","entry":"calculate_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":15,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":16,"n_samples_ran":7,"n_samples_fingerprinted":2,"n_places":17,"n_places_pointer_only":5,"by_status":{"ran_honours":3,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":4,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.17062","paper":"/paper/arxiv-2606-17062","title":"RadSEM: A Finding-by-Finding Metric for Clinical Consistency in Radiology Reports","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"jdh-algo/RadSEM","path":"step/step3.py","file_url":"https://github.com/jdh-algo/RadSEM/blob/HEAD/step/step3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"29b27aedb8af8403","mcp_get_code":{"code_sha256":"29b27aedb8af8403"}},{"arxiv_id":"2606.10528","paper":"/paper/arxiv-2606-10528","title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"lmarena/arena-hard-auto","path":"BenchBuilder/filter.py","file_url":"https://github.com/lmarena/arena-hard-auto/blob/HEAD/BenchBuilder/filter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"06750b8d972efe27","mcp_get_code":{"code_sha256":"06750b8d972efe27"}},{"arxiv_id":"2604.24470","paper":"/paper/arxiv-2604-24470","title":"Zero-shot Large Language Models for Automatic Readability Assessment","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"kinimod23/GRANT","path":"unsupervised/BERT/apply.py","file_url":"https://github.com/kinimod23/GRANT/blob/HEAD/unsupervised/BERT/apply.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4399c8d238487913","mcp_get_code":{"code_sha256":"4399c8d238487913"}},{"arxiv_id":"2604.15385","paper":"/paper/arxiv-2604-15385","title":"Prompt-Driven Code Summarization: A Systematic Literature Review","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"afia2023/prompt-engineering","path":"Prompt-Engineering_SLR/Filter/initial_Screening.py","file_url":"https://github.com/afia2023/prompt-engineering/blob/HEAD/Prompt-Engineering_SLR/Filter/initial_Screening.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"334a368192eaef1c","mcp_get_code":{"code_sha256":"334a368192eaef1c"}},{"arxiv_id":"2509.14232","paper":"/paper/arxiv-2509-14232","title":"GenExam: A Multidisciplinary Text-to-Image Exam","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"OpenGVLab/GenExam","path":"cal_score.py","file_url":"https://github.com/OpenGVLab/GenExam/blob/HEAD/cal_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fdc208489297bdd5","mcp_get_code":{"code_sha256":"fdc208489297bdd5"}},{"arxiv_id":"2507.23284","paper":null,"title":"arXiv:2507.23284","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"mlvlab/BLiM","path":"training_utils.py","file_url":"https://github.com/mlvlab/BLiM/blob/HEAD/training_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12161c0142e25620","mcp_get_code":{"code_sha256":"12161c0142e25620"}},{"arxiv_id":"2505.11765","paper":"/paper/omac-a-broad-optimization-framework-for-llm","title":"OMAC: A Broad Optimization Framework for LLM-Based Multi-Agent Collaboration","date":"2025-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiwenchao/OMAC","path":"code/MMLU/calc_score.py","file_url":"https://github.com/xiwenchao/OMAC/blob/HEAD/code/MMLU/calc_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7d74bc4eba65949","mcp_get_code":{"code_sha256":"a7d74bc4eba65949"}},{"arxiv_id":"2505.11765","paper":"/paper/omac-a-broad-optimization-framework-for-llm","title":"OMAC: A Broad Optimization Framework for LLM-Based Multi-Agent Collaboration","date":"2025-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiwenchao/OMAC","path":"code/HumanEval/utils_evo.py","file_url":"https://github.com/xiwenchao/OMAC/blob/HEAD/code/HumanEval/utils_evo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"23694908a81661df","mcp_get_code":{"code_sha256":"23694908a81661df"}},{"arxiv_id":"2505.08775","paper":"/paper/healthbench-evaluating-large-language-models","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","date":"2025-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openai/simple-evals","path":"healthbench_eval.py","file_url":"https://github.com/openai/simple-evals/blob/HEAD/healthbench_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c96e81709590b350","mcp_get_code":{"code_sha256":"c96e81709590b350"}},{"arxiv_id":"2503.07703","paper":"/paper/seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DYEvaLab/EvalMuse","path":"NTIRE2025/Track1_Alignment/evaluate.py","file_url":"https://github.com/DYEvaLab/EvalMuse/blob/HEAD/NTIRE2025/Track1_Alignment/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a33274f4811ba44b","mcp_get_code":{"code_sha256":"a33274f4811ba44b"}},{"arxiv_id":"2410.10502","paper":"/paper/a-practical-approach-to-causal-inference-over","title":"A Practical Approach to Causal Inference over Time","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"marti5ini/ci-over-time","path":"experiments/utils.py","file_url":"https://github.com/marti5ini/ci-over-time/blob/HEAD/experiments/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"e589f9c779df5ecf","mcp_get_code":{"code_sha256":"e589f9c779df5ecf"}},{"arxiv_id":"2406.11939","paper":"/paper/from-crowdsourced-data-to-high-quality","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/arena-hard","path":"BenchBuilder/filter.py","file_url":"https://github.com/lm-sys/arena-hard/blob/HEAD/BenchBuilder/filter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"06750b8d972efe27","mcp_get_code":{"code_sha256":"06750b8d972efe27"}},{"arxiv_id":"2405.03446","paper":"/paper/sevenllm-benchmarking-eliciting-and-enhancing","title":"SEvenLLM: Benchmarking, Eliciting, and Enhancing Abilities of Large Language Models in Cyber Threat Intelligence","date":null,"month_inferred_from_arxiv_id":"2024-05","title_source":"archive","repo":"csjianyang/seevenllm","path":"code/score/f1_rougel/ex-score-en-path.py","file_url":"https://github.com/csjianyang/seevenllm/blob/HEAD/code/score/f1_rougel/ex-score-en-path.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"532740bbf39ba1c0","mcp_get_code":{"code_sha256":"532740bbf39ba1c0"}},{"arxiv_id":"2405.03446","paper":"/paper/sevenllm-benchmarking-eliciting-and-enhancing","title":"SEvenLLM: Benchmarking, Eliciting, and Enhancing Abilities of Large Language Models in Cyber Threat Intelligence","date":null,"month_inferred_from_arxiv_id":"2024-05","title_source":"archive","repo":"csjianyang/seevenllm","path":"code/score/f1_rougel/gen_score_rougeL-en-path.py","file_url":"https://github.com/csjianyang/seevenllm/blob/HEAD/code/score/f1_rougel/gen_score_rougeL-en-path.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"593741de89b5746c","mcp_get_code":{"code_sha256":"593741de89b5746c"}},{"arxiv_id":"2309.16240","paper":"/paper/beyond-reverse-kl-generalizing-direct","title":"Beyond Reverse KL: Generalizing Direct Preference Optimization with Diverse Divergence Constraints","date":"2023-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alecwangcq/f-divergence-dpo","path":"cali/hh_cali_eval.py","file_url":"https://github.com/alecwangcq/f-divergence-dpo/blob/HEAD/cali/hh_cali_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f2074220eaa2a95d","mcp_get_code":{"code_sha256":"f2074220eaa2a95d"}},{"arxiv_id":"2308.16463","paper":"/paper/sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hypjudy/sparkles","path":"evaluate.py","file_url":"https://github.com/hypjudy/sparkles/blob/HEAD/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"8286c94254f1fdc5","mcp_get_code":{"code_sha256":"8286c94254f1fdc5"}},{"arxiv_id":"2022.findings-naacl.156","paper":null,"title":"arXiv:2022.findings-naacl.156","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"MJ-Jang/beyond-distributional","path":"src/analysis/wordvector_analysis.py","file_url":"https://github.com/MJ-Jang/beyond-distributional/blob/HEAD/src/analysis/wordvector_analysis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"3725cffc4daa4871","mcp_get_code":{"code_sha256":"3725cffc4daa4871"}}]}