{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/last-boxed-only-string","entry":"last_boxed_only_string","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":46,"n_papers_ran":38,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":20,"n_samples_ran":11,"n_samples_fingerprinted":11,"n_places":47,"n_places_pointer_only":23,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":8,"ran_fixture":0,"ran":3,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.11506","paper":"/paper/arxiv-2607-11506","title":"SCOPE-RL: Optimizing Reasoning Paths Before and After Success","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"tokencraft-lab/SCOPE-RL","path":"verl/recipe/scope_rl/reward_score/step_quality.py","file_url":"https://github.com/tokencraft-lab/SCOPE-RL/blob/HEAD/verl/recipe/scope_rl/reward_score/step_quality.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e76cb5ca340e2e17","mcp_get_code":{"code_sha256":"e76cb5ca340e2e17"}},{"arxiv_id":"2606.02684","paper":"/paper/arxiv-2606-02684","title":"Filter, Then Reweight: Rethinking Optimization Granularity in On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"YuYingLi0/FiRe-OPD","path":"math_eval/eval_math.py","file_url":"https://github.com/YuYingLi0/FiRe-OPD/blob/HEAD/math_eval/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2605.26971","paper":"/paper/arxiv-2605-26971","title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Celine-hxy/ATLAS","path":"verl/verl/utils/reward_score/math_dapo.py","file_url":"https://github.com/Celine-hxy/ATLAS/blob/HEAD/verl/verl/utils/reward_score/math_dapo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"885dcaee42326ed4","mcp_get_code":{"code_sha256":"885dcaee42326ed4"}},{"arxiv_id":"2605.26952","paper":"/paper/arxiv-2605-26952","title":"Efficient Agentic Reinforcement Learning with On-Policy Intrinsic Knowledge Boundary Enhancement","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"CuSO4-Chen/AKBE","path":"AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","file_url":"https://github.com/CuSO4-Chen/AKBE/blob/HEAD/AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a459ca4d242e275d","mcp_get_code":{"code_sha256":"a459ca4d242e275d"}},{"arxiv_id":"2605.11739","paper":"/paper/arxiv-2605-11739","title":"Learning to Foresee: Unveiling the Unlocking Efficiency of On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"caiyuchen-ustc/EffOPD","path":"EffOPD/math_eval/eval_math.py","file_url":"https://github.com/caiyuchen-ustc/EffOPD/blob/HEAD/EffOPD/math_eval/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2604.23626","paper":"/paper/arxiv-2604-23626","title":"GraphPlanner: Graph Memory-Augmented Agentic Routing for Multi-Agent LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ulab-uiuc/GraphPlanner","path":"router_planner/shared/math_eval.py","file_url":"https://github.com/ulab-uiuc/GraphPlanner/blob/HEAD/router_planner/shared/math_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9194e1f637041d86","mcp_get_code":{"code_sha256":"9194e1f637041d86"}},{"arxiv_id":"2604.18124","paper":"/paper/arxiv-2604-18124","title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rambo-Yi/TLora","path":"evaluate/math/util.py","file_url":"https://github.com/Rambo-Yi/TLora/blob/HEAD/evaluate/math/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/eval/eval_math.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/eval/eval_math.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0975c53e2a5fc2ab","mcp_get_code":{"code_sha256":"0975c53e2a5fc2ab"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/math_eval.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/math_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2602.22296","paper":"/paper/arxiv-2602-22296","title":"UpSkill: Mutual Information Skill Learning for Structured Response Diversity in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"dshah02/upskill","path":"src/DAPO_math_dapo.py","file_url":"https://github.com/dshah02/upskill/blob/HEAD/src/DAPO_math_dapo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"885dcaee42326ed4","mcp_get_code":{"code_sha256":"885dcaee42326ed4"}},{"arxiv_id":"2602.22296","paper":"/paper/arxiv-2602-22296","title":"UpSkill: Mutual Information Skill Learning for Structured Response Diversity in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"dshah02/upskill","path":"src/flex_extract.py","file_url":"https://github.com/dshah02/upskill/blob/HEAD/src/flex_extract.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8e1ba71a728e180c","mcp_get_code":{"code_sha256":"8e1ba71a728e180c"}},{"arxiv_id":"2602.19049","paper":"/paper/arxiv-2602-19049","title":"IAPO: Information-Aware Policy Optimization for Token-Efficient Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"YinhanHe123/IAPO","path":"compute_reward.py","file_url":"https://github.com/YinhanHe123/IAPO/blob/HEAD/compute_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"270295c5488675af","mcp_get_code":{"code_sha256":"270295c5488675af"}},{"arxiv_id":"2601.21484","paper":"/paper/arxiv-2601-21484","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"sheriyuo/ETS","path":"aime24/utils.py","file_url":"https://github.com/sheriyuo/ETS/blob/HEAD/aime24/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6d0bdd1050a5f9b9","mcp_get_code":{"code_sha256":"6d0bdd1050a5f9b9"}},{"arxiv_id":"2511.08043","paper":"/paper/arxiv-2511-08043","title":"DYNAACT: Large Language Model Reasoning with Dynamic Action Spaces","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"zhaoxlpku/DynaAct","path":"math_utils.py","file_url":"https://github.com/zhaoxlpku/DynaAct/blob/HEAD/math_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2507.06167","paper":"/paper/skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seephys/seephys-project","path":"vlmeval/api/bluelm_api.py","file_url":"https://github.com/seephys/seephys-project/blob/HEAD/vlmeval/api/bluelm_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a37f5e20fcdfc1ff","mcp_get_code":{"code_sha256":"a37f5e20fcdfc1ff"}},{"arxiv_id":"2506.24119","paper":"/paper/spiral-self-play-on-zero-sum-games","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","date":"2025-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spiral-rl/spiral","path":"spiral/utils.py","file_url":"https://github.com/spiral-rl/spiral/blob/HEAD/spiral/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2505.14625","paper":"/paper/tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uw-nsl/tinyv","path":"verl/verl/utils/reward_score/tinyv.py","file_url":"https://github.com/uw-nsl/tinyv/blob/HEAD/verl/verl/utils/reward_score/tinyv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b6bc0ef47c33095e","mcp_get_code":{"code_sha256":"b6bc0ef47c33095e"}},{"arxiv_id":"2505.10981","paper":"/paper/2505-10981","title":"Rethinking the Role of Prompting Strategies in LLM Test-Time Scaling: A Perspective of Probability Theory","date":"2025-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MraDonkey/rethinking_prompting","path":"model.py","file_url":"https://github.com/MraDonkey/rethinking_prompting/blob/HEAD/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2505.10518","paper":"/paper/multi-token-prediction-needs-registers","title":"Multi-Token Prediction Needs Registers","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nasosger/mutor","path":"language_modeling/src/eval/math_utils.py","file_url":"https://github.com/nasosger/mutor/blob/HEAD/language_modeling/src/eval/math_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2503.05188","paper":null,"title":"arXiv:2503.05188","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"BugMakerzzz/CRISP","path":"crisp_reason.py","file_url":"https://github.com/BugMakerzzz/CRISP/blob/HEAD/crisp_reason.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fbe5c4948905ca15","mcp_get_code":{"code_sha256":"fbe5c4948905ca15"}},{"arxiv_id":"2503.02390","paper":"/paper/reso-a-reward-driven-self-organizing-llm","title":"ReSo: A Reward-driven Self-organizing LLM-based Multi-Agent System for Reasoning Tasks","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":"archive","repo":"hengzzzhou/reso","path":"ReSo/agent_graph/agent_graph.py","file_url":"https://github.com/hengzzzhou/reso/blob/HEAD/ReSo/agent_graph/agent_graph.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"94659aac341e7312","mcp_get_code":{"code_sha256":"94659aac341e7312"}},{"arxiv_id":"2502.17387","paper":"/paper/big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"synthlabsai/big-math","path":"signals/rollouts_based_signals/math_eval.py","file_url":"https://github.com/synthlabsai/big-math/blob/HEAD/signals/rollouts_based_signals/math_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"abd7f82534e7b822","mcp_get_code":{"code_sha256":"abd7f82534e7b822"}},{"arxiv_id":"2502.14922","paper":"/paper/sift-grounding-llm-reasoning-in-contexts-via","title":"SIFT: Grounding LLM Reasoning in Contexts via Stickers","date":"2025-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhijie-group/sift","path":"acc_stage2.py","file_url":"https://github.com/zhijie-group/sift/blob/HEAD/acc_stage2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2502.12215","paper":"/paper/revisiting-the-test-time-scaling-of-o1-like","title":"Revisiting the Test-Time Scaling of o1-like Models: Do they Truly Possess Test-Time Scaling Capabilities?","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhiYuanZeng/test-time-scaling-eval","path":"math_evaluator.py","file_url":"https://github.com/ZhiYuanZeng/test-time-scaling-eval/blob/HEAD/math_evaluator.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2502.11133","paper":"/paper/masrouter-learning-to-route-llms-for-multi","title":"MasRouter: Learning to Route LLMs for Multi-Agent Systems","date":"2025-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yanweiyue/masrouter","path":"Datasets/math_dataset.py","file_url":"https://github.com/yanweiyue/masrouter/blob/HEAD/Datasets/math_dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2502.06781","paper":"/paper/exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"internlm/oreal","path":"oreal/judgers/utils.py","file_url":"https://github.com/internlm/oreal/blob/HEAD/oreal/judgers/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2411.07681","paper":"/paper/what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"katiekang1998/reasoning_generalization","path":"math_eval_samples.py","file_url":"https://github.com/katiekang1998/reasoning_generalization/blob/HEAD/math_eval_samples.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2410.15700","paper":"/paper/internlm2-5-stepprover-advancing-automated","title":"InternLM2.5-StepProver: Advancing Automated Theorem Proving via Expert Iteration on Large-Scale LEAN Problems","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"internlm/internlm-math","path":"agent/math_agent.py","file_url":"https://github.com/internlm/internlm-math/blob/HEAD/agent/math_agent.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2410.07627","paper":"/paper/automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/auto-cei","path":"data/MATH/data_pre.py","file_url":"https://github.com/salesforceairesearch/auto-cei/blob/HEAD/data/MATH/data_pre.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8529d921bdf3f1aa","mcp_get_code":{"code_sha256":"8529d921bdf3f1aa"}},{"arxiv_id":"2410.02884","paper":"/paper/llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2409.19734","paper":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nctu-eva-lab/vhd11k","path":"autogen/math_utils.py","file_url":"https://github.com/nctu-eva-lab/vhd11k/blob/HEAD/autogen/math_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"735fc5841569ed55","mcp_get_code":{"code_sha256":"735fc5841569ed55"}},{"arxiv_id":"2409.12147","paper":"/paper/magicore-multi-agent-iterative-coarse-to-fine","title":"MAgICoRe: Multi-Agent, Iterative, Coarse-to-Fine Refinement for Reasoning","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dinobby/magicore","path":"math_utils.py","file_url":"https://github.com/dinobby/magicore/blob/HEAD/math_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2408.04556","paper":"/paper/bias-aware-low-rank-adaptation-mitigating","title":"BA-LoRA: Bias-Alleviating Low-Rank Adaptation to Mitigate Catastrophic Inheritance in Large Language Models","date":"2024-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyp-jlu-ai/ba-lora","path":"inference/util.py","file_url":"https://github.com/cyp-jlu-ai/ba-lora/blob/HEAD/inference/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2406.14024","paper":"/paper/the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kbsdjames/math-minos","path":"evaluation/util.py","file_url":"https://github.com/kbsdjames/math-minos/blob/HEAD/evaluation/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2406.07394","paper":"/paper/accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trotsky1997/mathblackbox","path":"run_with_earlystopping.py","file_url":"https://github.com/trotsky1997/mathblackbox/blob/HEAD/run_with_earlystopping.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2405.15179","paper":"/paper/vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leo-yangli/VB-LoRA","path":"math_instruction_tuning/instruction_tuning_eval/utils.py","file_url":"https://github.com/leo-yangli/VB-LoRA/blob/HEAD/math_instruction_tuning/instruction_tuning_eval/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2405.06680","paper":"/paper/exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tongjingqi/MathTrap","path":"eval/util.py","file_url":"https://github.com/tongjingqi/MathTrap/blob/HEAD/eval/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2404.16811","paper":"/paper/make-your-llm-fully-utilize-the-context","title":"Make Your LLM Fully Utilize the Context","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/FILM","path":"short_tasks/utils.py","file_url":"https://github.com/microsoft/FILM/blob/HEAD/short_tasks/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37773130a24557b7","mcp_get_code":{"code_sha256":"37773130a24557b7"}},{"arxiv_id":"2404.02078","paper":"/paper/advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openbmb/eurus","path":"eval/utils/util.py","file_url":"https://github.com/openbmb/eurus/blob/HEAD/eval/utils/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2310.05914","paper":"/paper/neftune-noisy-embeddings-improve-instruction","title":"NEFTune: Noisy Embeddings Improve Instruction Finetuning","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akjindal53244/arithmo","path":"eval/MATH/MATH_compute_metric_zero_shot_CoT.py","file_url":"https://github.com/akjindal53244/arithmo/blob/HEAD/eval/MATH/MATH_compute_metric_zero_shot_CoT.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2309.06275","paper":"/paper/2309-06275","title":"Re-Reading Improves Reasoning in Large Language Models","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tebmer/rereading-llm-reasoning","path":"utils/util.py","file_url":"https://github.com/tebmer/rereading-llm-reasoning/blob/HEAD/utils/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2303.12712","paper":"/paper/sparks-of-artificial-general-intelligence","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"emrgnt-cmplxty/zero-shot-replication","path":"zero_shot_replication/core/math_helpers.py","file_url":"https://github.com/emrgnt-cmplxty/zero-shot-replication/blob/HEAD/zero_shot_replication/core/math_helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a6e611557b1300a9","mcp_get_code":{"code_sha256":"a6e611557b1300a9"}},{"arxiv_id":"2303.04673","paper":"/paper/cost-effective-hyperparameter-optimization","title":"Cost-Effective Hyperparameter Optimization for Large Language Model Generation Inference","date":"2023-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevin666aa/flaml","path":"flaml/autogen/math_utils.py","file_url":"https://github.com/kevin666aa/flaml/blob/HEAD/flaml/autogen/math_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"735fc5841569ed55","mcp_get_code":{"code_sha256":"735fc5841569ed55"}},{"arxiv_id":"openreview_QrC8OgQyOI","paper":null,"title":"arXiv:openreview_QrC8OgQyOI","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"viiika/Prism","path":"Dream/Dream_Prism/metrics/gsmk8_eval.py","file_url":"https://github.com/viiika/Prism/blob/HEAD/Dream/Dream_Prism/metrics/gsmk8_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a774929083345c9b","mcp_get_code":{"code_sha256":"a774929083345c9b"}},{"arxiv_id":"2025.findings-emnlp.598","paper":null,"title":"arXiv:2025.findings-emnlp.598","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"RUCAIBox/MMATH","path":"utils.py","file_url":"https://github.com/RUCAIBox/MMATH/blob/HEAD/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"90b5c896e5eaea5e","mcp_get_code":{"code_sha256":"90b5c896e5eaea5e"}},{"arxiv_id":"2025.emnlp-main.1660","paper":null,"title":"arXiv:2025.emnlp-main.1660","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"dinobby/MAgICoRe","path":"math_utils.py","file_url":"https://github.com/dinobby/MAgICoRe/blob/HEAD/math_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}},{"arxiv_id":"2024.findings-emnlp.407","paper":null,"title":"arXiv:2024.findings-emnlp.407","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"amayuelas/multi-agent-attack","path":"multiagent_debate/math_parsing.py","file_url":"https://github.com/amayuelas/multi-agent-attack/blob/HEAD/multiagent_debate/math_parsing.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b14c648516c38a7","mcp_get_code":{"code_sha256":"0b14c648516c38a7"}}]}