{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/should-allow-eval","entry":"should_allow_eval","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":14,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":2,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":14,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":0},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12243","paper":"/paper/arxiv-2609-12243","title":"Chopthin-Consensus Power Sampling: A Diversity-Preserving Approach to LLM Decoding","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"MinooAhmadii/chopthin-consensus-power-sampling","path":"ccps/graders/math_grader.py","file_url":"https://github.com/MinooAhmadii/chopthin-consensus-power-sampling/blob/HEAD/ccps/graders/math_grader.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2605.21856","paper":"/paper/arxiv-2605-21856","title":"The Illusion of Reasoning: Exposing Evasive Data Contamination in LLMs via Zero-CoT Truncation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Yifan-Lan/zero-cot-probe","path":"math_grade.py","file_url":"https://github.com/Yifan-Lan/zero-cot-probe/blob/HEAD/math_grade.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2605.10195","paper":"/paper/arxiv-2605-10195","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"PKU-SEC-Lab/SPEX","path":"evaluate/evaluate_utils/grader.py","file_url":"https://github.com/PKU-SEC-Lab/SPEX/blob/HEAD/evaluate/evaluate_utils/grader.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2604.11510","paper":"/paper/arxiv-2604-11510","title":"Policy Split: Incentivizing Dual-Mode Exploration in LLM Reinforcement with Dual-Mode Entropy Regularization","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"BITHLP/PolicySplit","path":"verl/utils/reward_score/openai_math_grade.py","file_url":"https://github.com/BITHLP/PolicySplit/blob/HEAD/verl/utils/reward_score/openai_math_grade.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/grading/grader.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/grading/grader.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2602.10048","paper":"/paper/arxiv-2602-10048","title":"Long Chain-of-Thought Compression via Fine-Grained Group Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Mr-XcHan/FGO","path":"src/open_r1/grader.py","file_url":"https://github.com/Mr-XcHan/FGO/blob/HEAD/src/open_r1/grader.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2510.07312","paper":"/paper/arxiv-2510-07312","title":"BOOTSTRAPPING LLMS TO REASON OVER LONGER HORIZONS VIA REINFORCEMENT LEARNING","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"AlesyaIvanova/h1","path":"math_utils.py","file_url":"https://github.com/AlesyaIvanova/h1/blob/HEAD/math_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2505.13417","paper":"/paper/adaptthink-reasoning-models-can-learn-when-to","title":"AdaptThink: Reasoning Models Can Learn When to Think","date":"2025-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-keg/adaptthink","path":"src/adapt_think_rm.py","file_url":"https://github.com/thu-keg/adaptthink/blob/HEAD/src/adapt_think_rm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2505.13379","paper":"/paper/thinkless-llm-learns-when-to-think","title":"Thinkless: LLM Learns When to Think","date":"2025-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vainf/thinkless","path":"deepscaler/rewards/math_utils/utils.py","file_url":"https://github.com/vainf/thinkless/blob/HEAD/deepscaler/rewards/math_utils/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2505.03335","paper":"/paper/absolute-zero-reinforced-self-play-reasoning","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","date":"2025-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LeapLabTHU/Absolute-Zero-Reasoner","path":"absolute_zero_reasoner/rewards/math_utils.py","file_url":"https://github.com/LeapLabTHU/Absolute-Zero-Reasoner/blob/HEAD/absolute_zero_reasoner/rewards/math_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2503.10460","paper":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Qihoo360/Light-R1","path":"deepscaler-release/deepscaler/rewards/math_utils/utils.py","file_url":"https://github.com/Qihoo360/Light-R1/blob/HEAD/deepscaler-release/deepscaler/rewards/math_utils/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2502.06703","paper":"/paper/can-1b-llm-surpass-405b-llm-rethinking","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RyanLiu112/compute-optimal-tts","path":"src/envs/MATH/verify_utils.py","file_url":"https://github.com/RyanLiu112/compute-optimal-tts/blob/HEAD/src/envs/MATH/verify_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}},{"arxiv_id":"2406.12809","paper":"/paper/can-large-language-models-always-solve-easy","title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QwenLM/ConsisEval","path":"math_check/grader.py","file_url":"https://github.com/QwenLM/ConsisEval/blob/HEAD/math_check/grader.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f5c7c44887cd8c88","mcp_get_code":{"code_sha256":"f5c7c44887cd8c88"}},{"arxiv_id":"2305.20050","paper":"/paper/let-s-verify-step-by-step-1","title":"Let's Verify Step by Step","date":"2023-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openai/prm800k","path":"prm800k/grading/grader.py","file_url":"https://github.com/openai/prm800k/blob/HEAD/prm800k/grading/grader.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"95f02454eddf082a","mcp_get_code":{"code_sha256":"95f02454eddf082a"}}]}