{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-final-answer","entry":"extract_final_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":9,"n_samples_fingerprinted":9,"n_places":14,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":6,"ran_fixture":0,"ran":3,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.01224","paper":"/paper/arxiv-2609-01224","title":"S$^2$Prune: Spatially Structured Visual Token Pruning for Multimodal Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"yuanyuanjia71-spec/S2Prune","path":"s2prune/metrics.py","file_url":"https://github.com/yuanyuanjia71-spec/S2Prune/blob/HEAD/s2prune/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bf2ece2a59a66b89","mcp_get_code":{"code_sha256":"bf2ece2a59a66b89"}},{"arxiv_id":"2608.30156","paper":"/paper/arxiv-2608-30156","title":"Reactivating Test-Time Scaling for Plane Geometry Problem Solving","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Jason8Kang/ReTTS-PGPS","path":"src/retts_pgp/utils/verifier_v1.py","file_url":"https://github.com/Jason8Kang/ReTTS-PGPS/blob/HEAD/src/retts_pgp/utils/verifier_v1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"02852f1cf8dd3d3b","mcp_get_code":{"code_sha256":"02852f1cf8dd3d3b"}},{"arxiv_id":"2608.15962","paper":"/paper/arxiv-2608-15962","title":"SEER: Long-Context Reasoning via Selective Visual-Text Compression","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"jiaweixu98/SEER","path":"seer/infer/api_base.py","file_url":"https://github.com/jiaweixu98/SEER/blob/HEAD/seer/infer/api_base.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bddb0535ff92827e","mcp_get_code":{"code_sha256":"bddb0535ff92827e"}},{"arxiv_id":"2606.31048","paper":"/paper/arxiv-2606-31048","title":"Knowledge Distillation from Large Reasoning Models to Compact Student Models: A Case Study on the John O'Bryan Mathematics Competition","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"TempGaurab/Distillation.John-O-Bryan","path":"rab.py","file_url":"https://github.com/TempGaurab/Distillation.John-O-Bryan/blob/HEAD/rab.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e58cca20690182df","mcp_get_code":{"code_sha256":"e58cca20690182df"}},{"arxiv_id":"2606.18850","paper":"/paper/arxiv-2606-18850","title":"ScholarSum: Student-Teacher Abstractive Summarization via Knowledge Graph Reasoning and Reflective Refinement","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Xiaoyu-Tao/ScholarSum","path":"src/papersum/pipeline.py","file_url":"https://github.com/Xiaoyu-Tao/ScholarSum/blob/HEAD/src/papersum/pipeline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6d46a27345390734","mcp_get_code":{"code_sha256":"6d46a27345390734"}},{"arxiv_id":"2605.26414","paper":"/paper/arxiv-2605-26414","title":"Reasoning, Code, or Both? How Large Language Models Handle Variations in Math Questions","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"masamodelkin/llm-robustness-code-execution","path":"src/evals/SBSC.py","file_url":"https://github.com/masamodelkin/llm-robustness-code-execution/blob/HEAD/src/evals/SBSC.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3f5c70d489db7cf6","mcp_get_code":{"code_sha256":"3f5c70d489db7cf6"}},{"arxiv_id":"2604.25039","paper":"/paper/arxiv-2604-25039","title":"Dual-Track CoT: Budget-Aware Stepwise Guidance for Small LMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"atharvadpatil/DualTrack-COT","path":"src/dataset-preparation/gsm8k_stepwise_dataset.py","file_url":"https://github.com/atharvadpatil/DualTrack-COT/blob/HEAD/src/dataset-preparation/gsm8k_stepwise_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"22b603c7b61b0ff0","mcp_get_code":{"code_sha256":"22b603c7b61b0ff0"}},{"arxiv_id":"2601.04537","paper":"/paper/arxiv-2601-04537","title":"Linear Dynamics in the RLVR Training of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Miaow-Lab/RLVR-Linearity","path":"evaluation/pass_at_k_eval.py","file_url":"https://github.com/Miaow-Lab/RLVR-Linearity/blob/HEAD/evaluation/pass_at_k_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"60425d0b541f8efd","mcp_get_code":{"code_sha256":"60425d0b541f8efd"}},{"arxiv_id":"2506.10822","paper":"/paper/recut-balancing-reasoning-length-and-accuracy","title":"ReCUT: Balancing Reasoning Length and Accuracy in LLMs via Stepwise Trails and Preference Optimization","date":"2025-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"neuir/recut","path":"src/long2short_reward_generate.py","file_url":"https://github.com/neuir/recut/blob/HEAD/src/long2short_reward_generate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8bc54889033168e8","mcp_get_code":{"code_sha256":"8bc54889033168e8"}},{"arxiv_id":"2506.08691","paper":"/paper/vrest-enhancing-reasoning-in-large-vision","title":"VReST: Enhancing Reasoning in Large Vision-Language Models through Tree Search and Self-Reward Mechanism","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GaryJiajia/VReST","path":"prompt_methods/ours/our6.py","file_url":"https://github.com/GaryJiajia/VReST/blob/HEAD/prompt_methods/ours/our6.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c2090c70720744a9","mcp_get_code":{"code_sha256":"c2090c70720744a9"}},{"arxiv_id":"2502.14815","paper":"/paper/optimizing-model-selection-for-compound-ai","title":"Optimizing Model Selection for Compound AI Systems","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LLMSELECTOR/LLMSELECTOR","path":"llmselector/llmselector/compoundai/diagnoser.py","file_url":"https://github.com/LLMSELECTOR/LLMSELECTOR/blob/HEAD/llmselector/llmselector/compoundai/diagnoser.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5672cff2de3b190b","mcp_get_code":{"code_sha256":"5672cff2de3b190b"}},{"arxiv_id":"2502.14634","paper":"/paper/cer-confidence-enhanced-reasoning-in-llms","title":"CER: Confidence Enhanced Reasoning in LLMs","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sharif-ml-lab/CER","path":"src/decoding.py","file_url":"https://github.com/sharif-ml-lab/CER/blob/HEAD/src/decoding.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"89e7e36459462915","mcp_get_code":{"code_sha256":"89e7e36459462915"}},{"arxiv_id":"2406.10785","paper":"/paper/sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Rain9876/ShareLoRA","path":"sharelora/eval_callback.py","file_url":"https://github.com/Rain9876/ShareLoRA/blob/HEAD/sharelora/eval_callback.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bb24d01bb1bce9c8","mcp_get_code":{"code_sha256":"bb24d01bb1bce9c8"}},{"arxiv_id":"2310.00280","paper":"/paper/corex-pushing-the-boundaries-of-complex","title":"Corex: Pushing the Boundaries of Complex Reasoning through Multi-Model Collaboration","date":"2023-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QiushiSun/Corex","path":"corex_discuss.py","file_url":"https://github.com/QiushiSun/Corex/blob/HEAD/corex_discuss.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f82497a980f43756","mcp_get_code":{"code_sha256":"f82497a980f43756"}}]}