{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/remove-boxed","entry":"remove_boxed","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":48,"n_papers_ran":38,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":11,"n_samples_fingerprinted":6,"n_places":49,"n_places_pointer_only":29,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":7,"ran_fixture":0,"ran":4,"unverified":10},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.11506","paper":"/paper/arxiv-2607-11506","title":"SCOPE-RL: Optimizing Reasoning Paths Before and After Success","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"tokencraft-lab/SCOPE-RL","path":"verl/recipe/scope_rl/reward_score/step_quality.py","file_url":"https://github.com/tokencraft-lab/SCOPE-RL/blob/HEAD/verl/recipe/scope_rl/reward_score/step_quality.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bd6200a962c0b2b3","mcp_get_code":{"code_sha256":"bd6200a962c0b2b3"}},{"arxiv_id":"2606.18195","paper":"/paper/arxiv-2606-18195","title":"Learning from the Self-future: On-policy Self-distillation for dLLMs","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"xingzhejun/d-OPSD","path":"d-opsd/math500_utils.py","file_url":"https://github.com/xingzhejun/d-OPSD/blob/HEAD/d-opsd/math500_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4e621edb2336a842","mcp_get_code":{"code_sha256":"4e621edb2336a842"}},{"arxiv_id":"2606.02684","paper":"/paper/arxiv-2606-02684","title":"Filter, Then Reweight: Rethinking Optimization Granularity in On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"YuYingLi0/FiRe-OPD","path":"math_eval/eval_math.py","file_url":"https://github.com/YuYingLi0/FiRe-OPD/blob/HEAD/math_eval/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2605.29398","paper":"/paper/arxiv-2605-29398","title":"GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"GaryBall/GDSD","path":"gdsd/utils/math500_utils.py","file_url":"https://github.com/GaryBall/GDSD/blob/HEAD/gdsd/utils/math500_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e4d5a8bc57ae4499","mcp_get_code":{"code_sha256":"e4d5a8bc57ae4499"}},{"arxiv_id":"2605.26971","paper":"/paper/arxiv-2605-26971","title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Celine-hxy/ATLAS","path":"verl/verl/utils/reward_score/math_dapo.py","file_url":"https://github.com/Celine-hxy/ATLAS/blob/HEAD/verl/verl/utils/reward_score/math_dapo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52adf1b5d14ad2a3","mcp_get_code":{"code_sha256":"52adf1b5d14ad2a3"}},{"arxiv_id":"2605.26952","paper":"/paper/arxiv-2605-26952","title":"Efficient Agentic Reinforcement Learning with On-Policy Intrinsic Knowledge Boundary Enhancement","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"CuSO4-Chen/AKBE","path":"AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","file_url":"https://github.com/CuSO4-Chen/AKBE/blob/HEAD/AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6271eeda6c120b7","mcp_get_code":{"code_sha256":"f6271eeda6c120b7"}},{"arxiv_id":"2605.11739","paper":"/paper/arxiv-2605-11739","title":"Learning to Foresee: Unveiling the Unlocking Efficiency of On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"caiyuchen-ustc/EffOPD","path":"EffOPD/math_eval/eval_math.py","file_url":"https://github.com/caiyuchen-ustc/EffOPD/blob/HEAD/EffOPD/math_eval/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2604.23626","paper":"/paper/arxiv-2604-23626","title":"GraphPlanner: Graph Memory-Augmented Agentic Routing for Multi-Agent LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ulab-uiuc/GraphPlanner","path":"router_planner/shared/math_eval.py","file_url":"https://github.com/ulab-uiuc/GraphPlanner/blob/HEAD/router_planner/shared/math_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c70fa8b9ac912e78","mcp_get_code":{"code_sha256":"c70fa8b9ac912e78"}},{"arxiv_id":"2604.18124","paper":"/paper/arxiv-2604-18124","title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rambo-Yi/TLora","path":"evaluate/math/eval_math.py","file_url":"https://github.com/Rambo-Yi/TLora/blob/HEAD/evaluate/math/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/eval/eval_math.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/eval/eval_math.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed70d70cd67847d0","mcp_get_code":{"code_sha256":"ed70d70cd67847d0"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/math_eval.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/math_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2602.22296","paper":"/paper/arxiv-2602-22296","title":"UpSkill: Mutual Information Skill Learning for Structured Response Diversity in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"dshah02/upskill","path":"src/DAPO_math_dapo.py","file_url":"https://github.com/dshah02/upskill/blob/HEAD/src/DAPO_math_dapo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52adf1b5d14ad2a3","mcp_get_code":{"code_sha256":"52adf1b5d14ad2a3"}},{"arxiv_id":"2602.22296","paper":"/paper/arxiv-2602-22296","title":"UpSkill: Mutual Information Skill Learning for Structured Response Diversity in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"dshah02/upskill","path":"src/flex_extract.py","file_url":"https://github.com/dshah02/upskill/blob/HEAD/src/flex_extract.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"937ef0071534906e","mcp_get_code":{"code_sha256":"937ef0071534906e"}},{"arxiv_id":"2602.19049","paper":"/paper/arxiv-2602-19049","title":"IAPO: Information-Aware Policy Optimization for Token-Efficient Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"YinhanHe123/IAPO","path":"compute_reward.py","file_url":"https://github.com/YinhanHe123/IAPO/blob/HEAD/compute_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea98a34b07fe1bfa","mcp_get_code":{"code_sha256":"ea98a34b07fe1bfa"}},{"arxiv_id":"2602.06462","paper":"/paper/arxiv-2602-06462","title":"Diffusion-State Policy Optimization for Masked Diffusion Language Models","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"dllm-reasoning/d1","path":"diffu-grpo/math500_utils.py","file_url":"https://github.com/dllm-reasoning/d1/blob/HEAD/diffu-grpo/math500_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4e621edb2336a842","mcp_get_code":{"code_sha256":"4e621edb2336a842"}},{"arxiv_id":"2601.21484","paper":"/paper/arxiv-2601-21484","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"sheriyuo/ETS","path":"aime24/utils.py","file_url":"https://github.com/sheriyuo/ETS/blob/HEAD/aime24/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52adf1b5d14ad2a3","mcp_get_code":{"code_sha256":"52adf1b5d14ad2a3"}},{"arxiv_id":"2511.13223","paper":"/paper/arxiv-2511-13223","title":"TokenSqueeze: Performance-Preserving Compression for Reasoning LLMs","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"zhangyx1122/TokenSqueeze","path":"utils/math500_verify.py","file_url":"https://github.com/zhangyx1122/TokenSqueeze/blob/HEAD/utils/math500_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"53a8a2e9d14cedfd","mcp_get_code":{"code_sha256":"53a8a2e9d14cedfd"}},{"arxiv_id":"2511.08043","paper":"/paper/arxiv-2511-08043","title":"DYNAACT: Large Language Model Reasoning with Dynamic Action Spaces","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"zhaoxlpku/DynaAct","path":"math_utils.py","file_url":"https://github.com/zhaoxlpku/DynaAct/blob/HEAD/math_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2507.06167","paper":"/paper/skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seephys/seephys-project","path":"vlmeval/api/bluelm_api.py","file_url":"https://github.com/seephys/seephys-project/blob/HEAD/vlmeval/api/bluelm_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e531d45668a6c715","mcp_get_code":{"code_sha256":"e531d45668a6c715"}},{"arxiv_id":"2506.24119","paper":"/paper/spiral-self-play-on-zero-sum-games","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning","date":"2025-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spiral-rl/spiral","path":"spiral/utils.py","file_url":"https://github.com/spiral-rl/spiral/blob/HEAD/spiral/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2505.14625","paper":"/paper/tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uw-nsl/tinyv","path":"analysis_tool/verl_reward_score/math.py","file_url":"https://github.com/uw-nsl/tinyv/blob/HEAD/analysis_tool/verl_reward_score/math.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"53a8a2e9d14cedfd","mcp_get_code":{"code_sha256":"53a8a2e9d14cedfd"}},{"arxiv_id":"2503.05188","paper":null,"title":"arXiv:2503.05188","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2503.02390","paper":"/paper/reso-a-reward-driven-self-organizing-llm","title":"ReSo: A Reward-driven Self-organizing LLM-based Multi-Agent System for Reasoning Tasks","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":"archive","repo":"hengzzzhou/reso","path":"ReSo/agent_graph/agent_graph.py","file_url":"https://github.com/hengzzzhou/reso/blob/HEAD/ReSo/agent_graph/agent_graph.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f1b63da3ae1c5a7","mcp_get_code":{"code_sha256":"8f1b63da3ae1c5a7"}},{"arxiv_id":"2502.17387","paper":"/paper/big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"synthlabsai/big-math","path":"signals/rollouts_based_signals/math_eval.py","file_url":"https://github.com/synthlabsai/big-math/blob/HEAD/signals/rollouts_based_signals/math_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1fa9e0d56b6ac064","mcp_get_code":{"code_sha256":"1fa9e0d56b6ac064"}},{"arxiv_id":"2502.12215","paper":"/paper/revisiting-the-test-time-scaling-of-o1-like","title":"Revisiting the Test-Time Scaling of o1-like Models: Do they Truly Possess Test-Time Scaling Capabilities?","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhiYuanZeng/test-time-scaling-eval","path":"math_evaluator.py","file_url":"https://github.com/ZhiYuanZeng/test-time-scaling-eval/blob/HEAD/math_evaluator.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2502.07154","paper":"/paper/rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allanraventos/refine","path":"refine/training/filter_samples.py","file_url":"https://github.com/allanraventos/refine/blob/HEAD/refine/training/filter_samples.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"45f1d7f7f2d4a1e5","mcp_get_code":{"code_sha256":"45f1d7f7f2d4a1e5"}},{"arxiv_id":"2411.07681","paper":"/paper/what-do-learning-dynamics-reveal-about","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","date":"2024-11-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"katiekang1998/reasoning_generalization","path":"math_eval_samples.py","file_url":"https://github.com/katiekang1998/reasoning_generalization/blob/HEAD/math_eval_samples.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2410.07627","paper":"/paper/automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/auto-cei","path":"data/MATH/data_pre.py","file_url":"https://github.com/salesforceairesearch/auto-cei/blob/HEAD/data/MATH/data_pre.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"05ccbbbc214e3ed4","mcp_get_code":{"code_sha256":"05ccbbbc214e3ed4"}},{"arxiv_id":"2410.02884","paper":"/paper/llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2409.19734","paper":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nctu-eva-lab/vhd11k","path":"autogen/math_utils.py","file_url":"https://github.com/nctu-eva-lab/vhd11k/blob/HEAD/autogen/math_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"7e08ca15778b791e","mcp_get_code":{"code_sha256":"7e08ca15778b791e"}},{"arxiv_id":"2409.12183","paper":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps","title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2408.04556","paper":"/paper/bias-aware-low-rank-adaptation-mitigating","title":"BA-LoRA: Bias-Alleviating Low-Rank Adaptation to Mitigate Catastrophic Inheritance in Large Language Models","date":"2024-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyp-jlu-ai/ba-lora","path":"inference/math_inference.py","file_url":"https://github.com/cyp-jlu-ai/ba-lora/blob/HEAD/inference/math_inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2406.14024","paper":"/paper/the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kbsdjames/math-minos","path":"evaluation/eval_math.py","file_url":"https://github.com/kbsdjames/math-minos/blob/HEAD/evaluation/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2406.08903","paper":"/paper/delta-come-training-free-delta-compression","title":"Delta-CoMe: Training-Free Delta-Compression with Mixed-Precision for Large Language Models","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2406.07394","paper":"/paper/accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trotsky1997/mathblackbox","path":"run_with_earlystopping.py","file_url":"https://github.com/trotsky1997/mathblackbox/blob/HEAD/run_with_earlystopping.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2405.15179","paper":"/paper/vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leo-yangli/VB-LoRA","path":"math_instruction_tuning/instruction_tuning_eval/MATH_eval.py","file_url":"https://github.com/leo-yangli/VB-LoRA/blob/HEAD/math_instruction_tuning/instruction_tuning_eval/MATH_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2405.14804","paper":"/paper/can-llms-solve-longer-math-word-problems","title":"Can LLMs Solve longer Math Word Problems Better?","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2405.06680","paper":"/paper/exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tongjingqi/MathTrap","path":"eval/eval_MATH_category2.py","file_url":"https://github.com/tongjingqi/MathTrap/blob/HEAD/eval/eval_MATH_category2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2404.16811","paper":"/paper/make-your-llm-fully-utilize-the-context","title":"Make Your LLM Fully Utilize the Context","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/FILM","path":"short_tasks/utils.py","file_url":"https://github.com/microsoft/FILM/blob/HEAD/short_tasks/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2310.20689","paper":"/paper/learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/lema","path":"inference_code/utils.py","file_url":"https://github.com/microsoft/lema/blob/HEAD/inference_code/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2310.05914","paper":"/paper/neftune-noisy-embeddings-improve-instruction","title":"NEFTune: Noisy Embeddings Improve Instruction Finetuning","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akjindal53244/arithmo","path":"eval/MATH/MATH_compute_metric_zero_shot_CoT.py","file_url":"https://github.com/akjindal53244/arithmo/blob/HEAD/eval/MATH/MATH_compute_metric_zero_shot_CoT.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2309.12284","paper":"/paper/metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meta-math/MetaMath","path":"eval_math.py","file_url":"https://github.com/meta-math/MetaMath/blob/HEAD/eval_math.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2309.06275","paper":"/paper/2309-06275","title":"Re-Reading Improves Reasoning in Large Language Models","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tebmer/rereading-llm-reasoning","path":"utils/util.py","file_url":"https://github.com/tebmer/rereading-llm-reasoning/blob/HEAD/utils/util.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"2303.12712","paper":"/paper/sparks-of-artificial-general-intelligence","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"emrgnt-cmplxty/zero-shot-replication","path":"zero_shot_replication/core/math_helpers.py","file_url":"https://github.com/emrgnt-cmplxty/zero-shot-replication/blob/HEAD/zero_shot_replication/core/math_helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7ef45e32b9ecc76b","mcp_get_code":{"code_sha256":"7ef45e32b9ecc76b"}},{"arxiv_id":"2303.04673","paper":"/paper/cost-effective-hyperparameter-optimization","title":"Cost-Effective Hyperparameter Optimization for Large Language Model Generation Inference","date":"2023-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevin666aa/flaml","path":"flaml/autogen/math_utils.py","file_url":"https://github.com/kevin666aa/flaml/blob/HEAD/flaml/autogen/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"33b20bbce665cc7a","mcp_get_code":{"code_sha256":"33b20bbce665cc7a"}},{"arxiv_id":"2103.03874","paper":"/paper/measuring-mathematical-problem-solving-with","title":"Measuring Mathematical Problem Solving With the MATH Dataset","date":"2021-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}},{"arxiv_id":"openreview_QrC8OgQyOI","paper":null,"title":"arXiv:openreview_QrC8OgQyOI","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"viiika/Prism","path":"Dream/Dream_Prism/metrics/gsmk8_eval.py","file_url":"https://github.com/viiika/Prism/blob/HEAD/Dream/Dream_Prism/metrics/gsmk8_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fa5d8fca12c5c0c8","mcp_get_code":{"code_sha256":"fa5d8fca12c5c0c8"}},{"arxiv_id":"2025.findings-emnlp.598","paper":null,"title":"arXiv:2025.findings-emnlp.598","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"RUCAIBox/MMATH","path":"utils.py","file_url":"https://github.com/RUCAIBox/MMATH/blob/HEAD/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9a85fff3911bdfa2","mcp_get_code":{"code_sha256":"9a85fff3911bdfa2"}},{"arxiv_id":"2024.findings-emnlp.407","paper":null,"title":"arXiv:2024.findings-emnlp.407","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"amayuelas/multi-agent-attack","path":"multiagent_debate/math_parsing.py","file_url":"https://github.com/amayuelas/multi-agent-attack/blob/HEAD/multiagent_debate/math_parsing.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3bbe264b05aadd3","mcp_get_code":{"code_sha256":"f3bbe264b05aadd3"}}]}