{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-boxed-answer","entry":"extract_boxed_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":15,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":18,"n_samples_ran":12,"n_samples_fingerprinted":12,"n_places":27,"n_places_pointer_only":13,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":9,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.27448","paper":"/paper/arxiv-2608-27448","title":"TTPO: Test-Time Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ZJU-REAL/TTPO","path":"TTPO/grpo_train.py","file_url":"https://github.com/ZJU-REAL/TTPO/blob/HEAD/TTPO/grpo_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"728111e00c43689f","mcp_get_code":{"code_sha256":"728111e00c43689f"}},{"arxiv_id":"2608.27448","paper":"/paper/arxiv-2608-27448","title":"TTPO: Test-Time Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ZJU-REAL/TTPO","path":"TTPO/ttpo_voting.py","file_url":"https://github.com/ZJU-REAL/TTPO/blob/HEAD/TTPO/ttpo_voting.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52eae8df46a34940","mcp_get_code":{"code_sha256":"52eae8df46a34940"}},{"arxiv_id":"2608.27448","paper":"/paper/arxiv-2608-27448","title":"TTPO: Test-Time Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ZJU-REAL/TTPO","path":"TTPO/eval/evaluate_math.py","file_url":"https://github.com/ZJU-REAL/TTPO/blob/HEAD/TTPO/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2608.09228","paper":"/paper/arxiv-2608-09228","title":"Privileged Solutions or Context-Induced Teacher Behavior? Dissecting On-Policy Self-Distillation","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"MBZUAI-reasoninglab/OP2SD","path":"eval/evaluate_math.py","file_url":"https://github.com/MBZUAI-reasoninglab/OP2SD/blob/HEAD/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2608.03467","paper":"/paper/arxiv-2608-03467","title":"When Correct Solutions Repeat: Rarity-Aware Credit Redistribution for GRPO","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO","path":"domain_config.py","file_url":"https://github.com/CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO/blob/HEAD/domain_config.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f249e8be89b9526c","mcp_get_code":{"code_sha256":"f249e8be89b9526c"}},{"arxiv_id":"2607.26627","paper":"/paper/arxiv-2607-26627","title":"Revisiting Lossy Verification in Speculative Decoding: Mechanisms, Trade-offs, and Failure Modes","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"ZhouYuxuanYX/Fast-HSD","path":"fast_hsd/benchmarks/_math_scoring.py","file_url":"https://github.com/ZhouYuxuanYX/Fast-HSD/blob/HEAD/fast_hsd/benchmarks/_math_scoring.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd23a3bbdde761c5","mcp_get_code":{"code_sha256":"cd23a3bbdde761c5"}},{"arxiv_id":"2607.14552","paper":"/paper/arxiv-2607-14552","title":"Answer-Conditioned Chains of Thought Degrade Verifiable-Reasoning Distillation in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"js-lee-AI/answer-leakage","path":"answer_leakage/matching.py","file_url":"https://github.com/js-lee-AI/answer-leakage/blob/HEAD/answer_leakage/matching.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"793c211f50002bc3","mcp_get_code":{"code_sha256":"793c211f50002bc3"}},{"arxiv_id":"2606.23104","paper":"/paper/arxiv-2606-23104","title":"ReNIO: Reweighting Negative Trajectory Importance for LLM On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"BDML-lab/ReNIO","path":"grpo_train.py","file_url":"https://github.com/BDML-lab/ReNIO/blob/HEAD/grpo_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"728111e00c43689f","mcp_get_code":{"code_sha256":"728111e00c43689f"}},{"arxiv_id":"2606.23104","paper":"/paper/arxiv-2606-23104","title":"ReNIO: Reweighting Negative Trajectory Importance for LLM On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"BDML-lab/ReNIO","path":"eval/evaluate_math.py","file_url":"https://github.com/BDML-lab/ReNIO/blob/HEAD/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2605.21606","paper":"/paper/arxiv-2605-21606","title":"When Are Teacher Tokens Reliable? Position-Weighted On-Policy Self-Distillation for Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"SaFo-Lab/PW-OPSD","path":"eval/evaluate_math.py","file_url":"https://github.com/SaFo-Lab/PW-OPSD/blob/HEAD/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2605.17672","paper":"/paper/arxiv-2605-17672","title":"Stop When Reasoning Converges: Semantic-Preserving Early Exit for Reasoning Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"giovanni-vaccarino/PUMA","path":"puma/extract_final_candidates.py","file_url":"https://github.com/giovanni-vaccarino/PUMA/blob/HEAD/puma/extract_final_candidates.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d534cfac750249cf","mcp_get_code":{"code_sha256":"d534cfac750249cf"}},{"arxiv_id":"2605.09548","paper":"/paper/arxiv-2605-09548","title":"Crosslingual On-Policy Self-Distillation for Multilingual Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"cisnlp/COPSD","path":"multilingual_grpo_train.py","file_url":"https://github.com/cisnlp/COPSD/blob/HEAD/multilingual_grpo_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90054f5de2b3ede3","mcp_get_code":{"code_sha256":"90054f5de2b3ede3"}},{"arxiv_id":"2605.09548","paper":"/paper/arxiv-2605-09548","title":"Crosslingual On-Policy Self-Distillation for Multilingual Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"cisnlp/COPSD","path":"african_langs_eval/evaluate_math.py","file_url":"https://github.com/cisnlp/COPSD/blob/HEAD/african_langs_eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"48da3f1fbb868b57","mcp_get_code":{"code_sha256":"48da3f1fbb868b57"}},{"arxiv_id":"2605.05040","paper":"/paper/arxiv-2605-05040","title":"Preference-Based Self-Distillation: Beyond KL Matching via Reward Regularization","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"LucasXinYu/PBSD","path":"eval/evaluate_math.py","file_url":"https://github.com/LucasXinYu/PBSD/blob/HEAD/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2605.05040","paper":"/paper/arxiv-2605-05040","title":"Preference-Based Self-Distillation: Beyond KL Matching via Reward Regularization","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"LucasXinYu/PBSD","path":"verify_utils.py","file_url":"https://github.com/LucasXinYu/PBSD/blob/HEAD/verify_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d1e73ab2175498fb","mcp_get_code":{"code_sha256":"d1e73ab2175498fb"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/math_eval.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/math_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"09b470ed75dcda56","mcp_get_code":{"code_sha256":"09b470ed75dcda56"}},{"arxiv_id":"2603.10624","paper":"/paper/arxiv-2603-10624","title":"Reinforcement Learning with Conditional Expectation Reward","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"changyi7231/CER","path":"recipe/cer/src/data_preparation.py","file_url":"https://github.com/changyi7231/CER/blob/HEAD/recipe/cer/src/data_preparation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"904946a1f147981f","mcp_get_code":{"code_sha256":"904946a1f147981f"}},{"arxiv_id":"2602.19672","paper":"/paper/arxiv-2602-19672","title":"SkillOrchestra: Learning to Route Agents via Skill Transfer","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"jiayuww/SkillOrchestra","path":"skillorchestra/eval/metrics.py","file_url":"https://github.com/jiayuww/SkillOrchestra/blob/HEAD/skillorchestra/eval/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4e5e88005347906a","mcp_get_code":{"code_sha256":"4e5e88005347906a"}},{"arxiv_id":"2601.18734","paper":"/paper/arxiv-2601-18734","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"siyan-zhao/OPSD","path":"grpo_train.py","file_url":"https://github.com/siyan-zhao/OPSD/blob/HEAD/grpo_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"728111e00c43689f","mcp_get_code":{"code_sha256":"728111e00c43689f"}},{"arxiv_id":"2601.18734","paper":"/paper/arxiv-2601-18734","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"siyan-zhao/OPSD","path":"eval/evaluate_math.py","file_url":"https://github.com/siyan-zhao/OPSD/blob/HEAD/eval/evaluate_math.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b1030fc6426aa14c","mcp_get_code":{"code_sha256":"b1030fc6426aa14c"}},{"arxiv_id":"2601.06022","paper":"/paper/arxiv-2601-06022","title":"AdaFuse: Adaptive Ensemble Decoding with Test-Time Scaling for LLMs","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"CCM0111/AdaFuse","path":"AdaFuse_two_models.py","file_url":"https://github.com/CCM0111/AdaFuse/blob/HEAD/AdaFuse_two_models.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"40e202a1ceb43583","mcp_get_code":{"code_sha256":"40e202a1ceb43583"}},{"arxiv_id":"2505.10518","paper":"/paper/multi-token-prediction-needs-registers","title":"Multi-Token Prediction Needs Registers","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nasosger/mutor","path":"language_modeling/src/eval/evaluate_gsm8k.py","file_url":"https://github.com/nasosger/mutor/blob/HEAD/language_modeling/src/eval/evaluate_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1c971191751f6f79","mcp_get_code":{"code_sha256":"1c971191751f6f79"}},{"arxiv_id":"2503.06580","paper":"/paper/agent-models-internalizing-chain-of-action","title":"Agent models: Internalizing Chain-of-Action Generation into Reasoning models","date":"2025-03-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adam-bjtu/autocoa","path":"evaluation/eval_retrieval.py","file_url":"https://github.com/adam-bjtu/autocoa/blob/HEAD/evaluation/eval_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6d99fd0ee8832cb2","mcp_get_code":{"code_sha256":"6d99fd0ee8832cb2"}},{"arxiv_id":"2502.12215","paper":"/paper/revisiting-the-test-time-scaling-of-o1-like","title":"Revisiting the Test-Time Scaling of o1-like Models: Do they Truly Possess Test-Time Scaling Capabilities?","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhiYuanZeng/test-time-scaling-eval","path":"math_evaluator.py","file_url":"https://github.com/ZhiYuanZeng/test-time-scaling-eval/blob/HEAD/math_evaluator.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"841780ad150562ae","mcp_get_code":{"code_sha256":"841780ad150562ae"}},{"arxiv_id":"2410.02884","paper":"/paper/llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"841780ad150562ae","mcp_get_code":{"code_sha256":"841780ad150562ae"}},{"arxiv_id":"2406.07394","paper":"/paper/accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trotsky1997/mathblackbox","path":"run_with_earlystopping.py","file_url":"https://github.com/trotsky1997/mathblackbox/blob/HEAD/run_with_earlystopping.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"841780ad150562ae","mcp_get_code":{"code_sha256":"841780ad150562ae"}},{"arxiv_id":"2025.findings-acl.1225","paper":null,"title":"arXiv:2025.findings-acl.1225","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HieuNT91/attention_pruning","path":"src/math500_utils.py","file_url":"https://github.com/HieuNT91/attention_pruning/blob/HEAD/src/math500_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fb8a8b4f24c96ca5","mcp_get_code":{"code_sha256":"fb8a8b4f24c96ca5"}}]}