{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/check-correctness","entry":"check_correctness","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":26,"n_papers_ran":8,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":24,"n_samples_ran":8,"n_samples_fingerprinted":1,"n_places":26,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":7,"unverified":16},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12243","paper":"/paper/arxiv-2609-12243","title":"Chopthin-Consensus Power Sampling: A Diversity-Preserving Approach to LLM Decoding","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"MinooAhmadii/chopthin-consensus-power-sampling","path":"ccps/graders/he_execute.py","file_url":"https://github.com/MinooAhmadii/chopthin-consensus-power-sampling/blob/HEAD/ccps/graders/he_execute.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"fe90a22233c11824","mcp_get_code":{"code_sha256":"fe90a22233c11824"}},{"arxiv_id":"2609.02264","paper":"/paper/arxiv-2609-02264","title":"Codebook Agent: Amortized Topology Design for LLM Multi-Agent Systems","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"jinxiy1104/CodebookAgent","path":"benchmarks/humaneval_dataset.py","file_url":"https://github.com/jinxiy1104/CodebookAgent/blob/HEAD/benchmarks/humaneval_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2b71c1db65fa88c","mcp_get_code":{"code_sha256":"a2b71c1db65fa88c"}},{"arxiv_id":"2606.05859","paper":"/paper/arxiv-2606-05859","title":"TARPO: Token-Wise Latent-Explicit Reasoning via Action-Routing Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"NKU-LITI/TARPO-master","path":"humanevaleval.py","file_url":"https://github.com/NKU-LITI/TARPO-master/blob/HEAD/humanevaleval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1177382634e74969","mcp_get_code":{"code_sha256":"1177382634e74969"}},{"arxiv_id":"2605.23315","paper":"/paper/arxiv-2605-23315","title":"Convergence Without Understanding: When Language Models Agree on Representations but Disagree on Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Usama1002/convergence-without-understanding","path":"src/evaluation.py","file_url":"https://github.com/Usama1002/convergence-without-understanding/blob/HEAD/src/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42b68d72beb0cfc1","mcp_get_code":{"code_sha256":"42b68d72beb0cfc1"}},{"arxiv_id":"2605.12652","paper":"/paper/arxiv-2605-12652","title":"Multi-Rollout On-Policy Distillation via Peer Successes and Failures","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"viviable/mopd_code","path":"verl/utils/reward_score/prime_code/utils.py","file_url":"https://github.com/viviable/mopd_code/blob/HEAD/verl/utils/reward_score/prime_code/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2d94e42b971ca757","mcp_get_code":{"code_sha256":"2d94e42b971ca757"}},{"arxiv_id":"2604.12268","paper":"/paper/arxiv-2604-12268","title":"CODESPECBENCH: Benchmarking LLMs for Executable Behavioral Specification Generation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SparksofAGI/CodeSpecBench","path":"CodeSpecBench-Func/spec-verifier/evaluation.py","file_url":"https://github.com/SparksofAGI/CodeSpecBench/blob/HEAD/CodeSpecBench-Func/spec-verifier/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f623a2fab794092","mcp_get_code":{"code_sha256":"8f623a2fab794092"}},{"arxiv_id":"2602.13773","paper":"/paper/arxiv-2602-13773","title":"On Representation Redundancy in Large-Scale Instruction Tuning Data Selection","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"tdano1/CRDS","path":"main/eval/codex_humaneval/execution.py","file_url":"https://github.com/tdano1/CRDS/blob/HEAD/main/eval/codex_humaneval/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"006afab0fbb96716","mcp_get_code":{"code_sha256":"006afab0fbb96716"}},{"arxiv_id":"2602.11639","paper":"/paper/arxiv-2602-11639","title":"PACE: Prefix-Protected and Difficulty-Aware Compression for Efficient Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"LiveCodeBench/LiveCodeBench","path":"lcb_runner/evaluation/utils_execute.py","file_url":"https://github.com/LiveCodeBench/LiveCodeBench/blob/HEAD/lcb_runner/evaluation/utils_execute.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d4af0384e850be8d","mcp_get_code":{"code_sha256":"d4af0384e850be8d"}},{"arxiv_id":"2602.02979","paper":"/paper/arxiv-2602-02979","title":"CPMöbius: Iterative Coach-Player Reasoning for Data-Free Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"thunlp/CPMobius","path":"evaluation/utils/execution.py","file_url":"https://github.com/thunlp/CPMobius/blob/HEAD/evaluation/utils/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"12d4fef547070f7c","mcp_get_code":{"code_sha256":"12d4fef547070f7c"}},{"arxiv_id":"2602.01365","paper":"/paper/arxiv-2602-01365","title":"When Domains Interact: Asymmetric and Order-Sensitive Cross-Domain Effects in Reinforcement Learning for Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"uservan/cross_domain","path":"verify/score/science.py","file_url":"https://github.com/uservan/cross_domain/blob/HEAD/verify/score/science.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8894398260de443f","mcp_get_code":{"code_sha256":"8894398260de443f"}},{"arxiv_id":"2505.05315","paper":"/paper/scalable-chain-of-thoughts-via-elastic","title":"Scalable Chain of Thoughts via Elastic Reasoning","date":"2025-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/elastic-reasoning","path":"rllm/rewards/code_reward.py","file_url":"https://github.com/salesforceairesearch/elastic-reasoning/blob/HEAD/rllm/rewards/code_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"7134e7134bb2121a","mcp_get_code":{"code_sha256":"7134e7134bb2121a"}},{"arxiv_id":"2410.12381","paper":"/paper/humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HumanEval-V/HumanEval-V-Benchmark","path":"execution.py","file_url":"https://github.com/HumanEval-V/HumanEval-V-Benchmark/blob/HEAD/execution.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f0c3af44403d4d56","mcp_get_code":{"code_sha256":"f0c3af44403d4d56"}},{"arxiv_id":"2407.01910","paper":"/paper/mg-verilog-multi-grained-dataset-towards","title":"MG-Verilog: Multi-grained Dataset Towards Enhanced LLM-assisted Verilog Generation","date":"2024-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luke-avionics/mg-verilog","path":"verilog_eval/verilog_eval/execution.py","file_url":"https://github.com/luke-avionics/mg-verilog/blob/HEAD/verilog_eval/verilog_eval/execution.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f327928b4d07faf0","mcp_get_code":{"code_sha256":"f327928b4d07faf0"}},{"arxiv_id":"2406.18532","paper":"/paper/symbolic-learning-enables-self-evolving","title":"Symbolic Learning Enables Self-Evolving Agents","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aiwaves-cn/agents","path":"src/agents/datasets/humaneval.py","file_url":"https://github.com/aiwaves-cn/agents/blob/HEAD/src/agents/datasets/humaneval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ac8c3678ce38802","mcp_get_code":{"code_sha256":"0ac8c3678ce38802"}},{"arxiv_id":"2406.04531","paper":"/paper/testeval-benchmarking-large-language-models","title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","date":null,"month_inferred_from_arxiv_id":"2024-06","title_source":"archive","repo":"llm4softwaretesting/testeval","path":"eval_overall.py","file_url":"https://github.com/llm4softwaretesting/testeval/blob/HEAD/eval_overall.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"14f35d21477c79fd","mcp_get_code":{"code_sha256":"14f35d21477c79fd"}},{"arxiv_id":"2404.02078","paper":"/paper/advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openbmb/eurus","path":"eval/utils/execution.py","file_url":"https://github.com/openbmb/eurus/blob/HEAD/eval/utils/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"12d4fef547070f7c","mcp_get_code":{"code_sha256":"12d4fef547070f7c"}},{"arxiv_id":"2402.14852","paper":"/paper/humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"daniel442li/gpt-human-eval","path":"human_eval/execution.py","file_url":"https://github.com/daniel442li/gpt-human-eval/blob/HEAD/human_eval/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8084d416685d852c","mcp_get_code":{"code_sha256":"8084d416685d852c"}},{"arxiv_id":"2401.16405","paper":"/paper/scaling-sparse-fine-tuning-to-large-language","title":"Scaling Sparse Fine-Tuning to Large Language Models","date":"2024-01-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ducdauge/sft-llm","path":"eval/codex_humaneval/execution.py","file_url":"https://github.com/ducdauge/sft-llm/blob/HEAD/eval/codex_humaneval/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12d4fef547070f7c","mcp_get_code":{"code_sha256":"12d4fef547070f7c"}},{"arxiv_id":"2401.09395","paper":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"declare-lab/llm-reasoningtest","path":"execution.py","file_url":"https://github.com/declare-lab/llm-reasoningtest/blob/HEAD/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ba731bb9824c9e75","mcp_get_code":{"code_sha256":"ba731bb9824c9e75"}},{"arxiv_id":"2401.06628","paper":"/paper/oop-object-oriented-programming-evaluation","title":"OOP: Object-Oriented Programming Evaluation Benchmark for Large Language Models","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alphadl/oop-eval","path":"oop_evaluate/execution.py","file_url":"https://github.com/alphadl/oop-eval/blob/HEAD/oop_evaluate/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"669dc622124bea08","mcp_get_code":{"code_sha256":"669dc622124bea08"}},{"arxiv_id":"2401.03003","paper":"/paper/ast-t5-structure-aware-pretraining-for-code","title":"AST-T5: Structure-Aware Pretraining for Code Generation and Understanding","date":"2024-01-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gonglinyuan/ast_t5","path":"eval/evaluate_mbpp.py","file_url":"https://github.com/gonglinyuan/ast_t5/blob/HEAD/eval/evaluate_mbpp.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a1d883bd8569073","mcp_get_code":{"code_sha256":"9a1d883bd8569073"}},{"arxiv_id":"2310.20329","paper":"/paper/instructcoder-empowering-language-models-for","title":"InstructCoder: Instruction Tuning Large Language Models for Code Editing","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qishenghu/CodeInstruct","path":"edit_eval/execution.py","file_url":"https://github.com/qishenghu/CodeInstruct/blob/HEAD/edit_eval/execution.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d76cd6028267121b","mcp_get_code":{"code_sha256":"d76cd6028267121b"}},{"arxiv_id":"2310.06830","paper":"/paper/lemur-harmonizing-natural-language-and-code","title":"Lemur: Harmonizing Natural Language and Code for Language Agents","date":"2023-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OpenLemur/Lemur","path":"xchat/eval/mbpp/execution.py","file_url":"https://github.com/OpenLemur/Lemur/blob/HEAD/xchat/eval/mbpp/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b8cf46548e4cd074","mcp_get_code":{"code_sha256":"b8cf46548e4cd074"}},{"arxiv_id":"2207.10397","paper":"/paper/codet-code-generation-with-generated-tests","title":"CodeT: Code Generation with Generated Tests","date":"2022-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/codet","path":"CodeT/src/_execution.py","file_url":"https://github.com/microsoft/codet/blob/HEAD/CodeT/src/_execution.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4b2cbd2e86925044","mcp_get_code":{"code_sha256":"4b2cbd2e86925044"}},{"arxiv_id":"2202.01875","paper":"/paper/rethinking-explainability-as-a-dialogue-a","title":"Rethinking Explainability as a Dialogue: A Practitioner's Perspective","date":"2022-02-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dylan-slack/talktomodel","path":"experiments/utils.py","file_url":"https://github.com/dylan-slack/talktomodel/blob/HEAD/experiments/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b20ffd4960370230","mcp_get_code":{"code_sha256":"b20ffd4960370230"}},{"arxiv_id":"2107.03374","paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openai/human-eval","path":"human_eval/execution.py","file_url":"https://github.com/openai/human-eval/blob/HEAD/human_eval/execution.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"87135198de7ec14b","mcp_get_code":{"code_sha256":"87135198de7ec14b"}}]}