{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-code","entry":"extract_code","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":24,"n_papers_ran":16,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":27,"n_samples_ran":18,"n_samples_fingerprinted":17,"n_places":28,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":15,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.28166","paper":"/paper/arxiv-2607-28166","title":"Commit Locally, Exit Globally: Coordinating Adaptive Sampling and Early Exit in Diffusion Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"ming053l/C4-dLLM","path":"c4/codegen_grade.py","file_url":"https://github.com/ming053l/C4-dLLM/blob/HEAD/c4/codegen_grade.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1660493478fd2823","mcp_get_code":{"code_sha256":"1660493478fd2823"}},{"arxiv_id":"2607.26627","paper":"/paper/arxiv-2607-26627","title":"Revisiting Lossy Verification in Speculative Decoding: Mechanisms, Trade-offs, and Failure Modes","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"ZhouYuxuanYX/Fast-HSD","path":"fast_hsd/benchmarks/_mbppplus_scoring.py","file_url":"https://github.com/ZhouYuxuanYX/Fast-HSD/blob/HEAD/fast_hsd/benchmarks/_mbppplus_scoring.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6802b1b27591fad2","mcp_get_code":{"code_sha256":"6802b1b27591fad2"}},{"arxiv_id":"2606.25832","paper":"/paper/arxiv-2606-25832","title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Hsiang-1/MiniOpt","path":"rl/opt_reward.py","file_url":"https://github.com/Hsiang-1/MiniOpt/blob/HEAD/rl/opt_reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dd14950a6b76ffbc","mcp_get_code":{"code_sha256":"dd14950a6b76ffbc"}},{"arxiv_id":"2605.29398","paper":"/paper/arxiv-2605-29398","title":"GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"GaryBall/GDSD","path":"gdsd/rewards.py","file_url":"https://github.com/GaryBall/GDSD/blob/HEAD/gdsd/rewards.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0f07a2dc4671890b","mcp_get_code":{"code_sha256":"0f07a2dc4671890b"}},{"arxiv_id":"2605.26414","paper":"/paper/arxiv-2605-26414","title":"Reasoning, Code, or Both? How Large Language Models Handle Variations in Math Questions","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"masamodelkin/llm-robustness-code-execution","path":"src/evals/PAL.py","file_url":"https://github.com/masamodelkin/llm-robustness-code-execution/blob/HEAD/src/evals/PAL.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1b2e245d59818293","mcp_get_code":{"code_sha256":"1b2e245d59818293"}},{"arxiv_id":"2605.18374","paper":"/paper/arxiv-2605-18374","title":"Beyond Inference-Time Search: Reinforcement Learning Synthesizes Reusable Solvers","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"IDEALLab/neural-solver-synthesis","path":"evaluation/sds/universal_solver_search.py","file_url":"https://github.com/IDEALLab/neural-solver-synthesis/blob/HEAD/evaluation/sds/universal_solver_search.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5a3b15c8fcdee09d","mcp_get_code":{"code_sha256":"5a3b15c8fcdee09d"}},{"arxiv_id":"2605.02545","paper":"/paper/arxiv-2605-02545","title":"Strategy-Aware Optimization Modeling with Reasoning LLMs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"rachhhhing/SAGE","path":"eval/execute.py","file_url":"https://github.com/rachhhhing/SAGE/blob/HEAD/eval/execute.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dc0fc511d2aa9942","mcp_get_code":{"code_sha256":"dc0fc511d2aa9942"}},{"arxiv_id":"2604.07937","paper":"/paper/arxiv-2604-07937","title":"HCRE: LLM-based Hierarchical Classification for Cross-Document Relation Extraction with a Prediction-then-Verification Strategy","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"XMUDeepLIT/HCRE","path":"auto-tree/utils.py","file_url":"https://github.com/XMUDeepLIT/HCRE/blob/HEAD/auto-tree/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"495dba1ca2e724dc","mcp_get_code":{"code_sha256":"495dba1ca2e724dc"}},{"arxiv_id":"2604.00344","paper":"/paper/arxiv-2604-00344","title":"Agent Q-Mix: Selecting the Right Action for LLM Multi-Agent Systems through Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ericjiang18/Agent-Q-Mix","path":"agent_baseline/utils/code_extract.py","file_url":"https://github.com/ericjiang18/Agent-Q-Mix/blob/HEAD/agent_baseline/utils/code_extract.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e7afca0a2b453f4c","mcp_get_code":{"code_sha256":"e7afca0a2b453f4c"}},{"arxiv_id":"2505.20881","paper":"/paper/generalizable-heuristic-generation-through","title":"Generalizable Heuristic Generation Through Large Language Models with Meta-Optimization","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiding-s/MoH","path":"moh.py","file_url":"https://github.com/yiding-s/MoH/blob/HEAD/moh.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"02ffbc411e44afd8","mcp_get_code":{"code_sha256":"02ffbc411e44afd8"}},{"arxiv_id":"2505.03335","paper":"/paper/absolute-zero-reinforced-self-play-reasoning","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","date":"2025-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LeapLabTHU/Absolute-Zero-Reasoner","path":"absolute_zero_reasoner/rewards/code_reward.py","file_url":"https://github.com/LeapLabTHU/Absolute-Zero-Reasoner/blob/HEAD/absolute_zero_reasoner/rewards/code_reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dd14950a6b76ffbc","mcp_get_code":{"code_sha256":"dd14950a6b76ffbc"}},{"arxiv_id":"2505.03335","paper":"/paper/absolute-zero-reinforced-self-play-reasoning","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","date":"2025-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LeapLabTHU/Absolute-Zero-Reasoner","path":"absolute_zero_reasoner/rewards/custom_evaluate.py","file_url":"https://github.com/LeapLabTHU/Absolute-Zero-Reasoner/blob/HEAD/absolute_zero_reasoner/rewards/custom_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72c1465a54867eee","mcp_get_code":{"code_sha256":"72c1465a54867eee"}},{"arxiv_id":"2503.07358","paper":"/paper/repost-scalable-repository-level-coding","title":"RepoST: Scalable Repository-Level Coding Environment Construction with Sandbox Testing","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiqingxyq/RepoST","path":"utils.py","file_url":"https://github.com/yiqingxyq/RepoST/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"71210ee33ca4b9bb","mcp_get_code":{"code_sha256":"71210ee33ca4b9bb"}},{"arxiv_id":"2502.18581","paper":"/paper/scalable-best-of-n-selection-for-large","title":"Scalable Best-of-N Selection for Large Language Models via Self-Certainty","date":"2025-02-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"backprop07/Self-Certainty","path":"src/livecode_self_certainty_from_list.py","file_url":"https://github.com/backprop07/Self-Certainty/blob/HEAD/src/livecode_self_certainty_from_list.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8a7e4da9680173bb","mcp_get_code":{"code_sha256":"8a7e4da9680173bb"}},{"arxiv_id":"2409.19667","paper":"/paper/can-large-language-models-analyze-graphs-like","title":"Can Large Language Models Analyze Graphs like Professionals? A Benchmark, Datasets and Models","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bupt-gamma/graphteam","path":"multi-agents-4-graph-analysis/GraphTeam/camel/agents.py","file_url":"https://github.com/bupt-gamma/graphteam/blob/HEAD/multi-agents-4-graph-analysis/GraphTeam/camel/agents.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77d746c4da984457","mcp_get_code":{"code_sha256":"77d746c4da984457"}},{"arxiv_id":"2407.19633","paper":"/paper/optimus-0-3-using-large-language-models-to","title":"OptiMUS-0.3: Using Large Language Models to Model and Solve Optimization Problems at Scale","date":"2024-07-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teshnizi/optimus","path":"Reflexion.py","file_url":"https://github.com/teshnizi/optimus/blob/HEAD/Reflexion.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3400042ef10a04a1","mcp_get_code":{"code_sha256":"3400042ef10a04a1"}},{"arxiv_id":"2407.19633","paper":"/paper/optimus-0-3-using-large-language-models-to","title":"OptiMUS-0.3: Using Large Language Models to Model and Solve Optimization Problems at Scale","date":"2024-07-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teshnizi/optimus","path":"execute_code.py","file_url":"https://github.com/teshnizi/optimus/blob/HEAD/execute_code.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1b35aeee4688c891","mcp_get_code":{"code_sha256":"1b35aeee4688c891"}},{"arxiv_id":"2407.05700","paper":"/paper/inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wyt2000/InverseCoder","path":"src/InstGen/sample_vllm.py","file_url":"https://github.com/wyt2000/InverseCoder/blob/HEAD/src/InstGen/sample_vllm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4ba4d218cc21b7c3","mcp_get_code":{"code_sha256":"4ba4d218cc21b7c3"}},{"arxiv_id":"2407.05700","paper":"/paper/inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wyt2000/InverseCoder","path":"src/InstGen/sample_vllm_parallel.py","file_url":"https://github.com/wyt2000/InverseCoder/blob/HEAD/src/InstGen/sample_vllm_parallel.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e57c27d5464d633","mcp_get_code":{"code_sha256":"2e57c27d5464d633"}},{"arxiv_id":"2406.12793","paper":"/paper/chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/chatglm","path":"composite_demo/demo_tool.py","file_url":"https://github.com/thudm/chatglm/blob/HEAD/composite_demo/demo_tool.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e4671affec558222","mcp_get_code":{"code_sha256":"e4671affec558222"}},{"arxiv_id":"2405.14918","paper":"/paper/analogcoder-analog-circuit-design-via","title":"AnalogCoder: Analog Circuit Design via Training-Free Code Generation","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"laiyao1/AnalogCoder","path":"gpt_run.py","file_url":"https://github.com/laiyao1/AnalogCoder/blob/HEAD/gpt_run.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3248e0669f8d0053","mcp_get_code":{"code_sha256":"3248e0669f8d0053"}},{"arxiv_id":"2404.08877","paper":"/paper/aligning-llms-for-fl-free-program-repair","title":"Aligning the Objective of LLM-based Program Repair","date":"2024-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cuhk-shenzhen-se/d4c","path":"utils/patch_apply.py","file_url":"https://github.com/cuhk-shenzhen-se/d4c/blob/HEAD/utils/patch_apply.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d11e7e729a9dd4ea","mcp_get_code":{"code_sha256":"d11e7e729a9dd4ea"}},{"arxiv_id":"2403.05307","paper":"/paper/tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tapilot-crossing/tapilot_code","path":"postprocessing/combine_code_gen_AIR.py","file_url":"https://github.com/tapilot-crossing/tapilot_code/blob/HEAD/postprocessing/combine_code_gen_AIR.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c805463b7efcc6de","mcp_get_code":{"code_sha256":"c805463b7efcc6de"}},{"arxiv_id":"2403.05307","paper":"/paper/tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tapilot-crossing/tapilot_code","path":"postprocessing/combine_code_gen_base.py","file_url":"https://github.com/tapilot-crossing/tapilot_code/blob/HEAD/postprocessing/combine_code_gen_base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d2c5937f51ac5ab","mcp_get_code":{"code_sha256":"0d2c5937f51ac5ab"}},{"arxiv_id":"2403.02528","paper":"/paper/daco-towards-application-driven-and","title":"DACO: Towards Application-Driven and Comprehensive Data Analysis via Code Generation","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shirley-wu/daco","path":"code/utils.py","file_url":"https://github.com/shirley-wu/daco/blob/HEAD/code/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"723183a95fe27709","mcp_get_code":{"code_sha256":"723183a95fe27709"}},{"arxiv_id":"2310.02304","paper":"/paper/self-taught-optimizer-stop-recursively-self","title":"Self-Taught Optimizer (STOP): Recursively Self-Improving Code Generation","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/stop","path":"helpers.py","file_url":"https://github.com/microsoft/stop/blob/HEAD/helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc62532cec8323f5","mcp_get_code":{"code_sha256":"dc62532cec8323f5"}},{"arxiv_id":"2310.01361","paper":"/paper/gensim-generating-robotic-simulation-tasks","title":"GenSim: Generating Robotic Simulation Tasks via Large Language Models","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liruiw/gensim","path":"gensim/utils.py","file_url":"https://github.com/liruiw/gensim/blob/HEAD/gensim/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5f7af117975e9a6e","mcp_get_code":{"code_sha256":"5f7af117975e9a6e"}},{"arxiv_id":"2303.04673","paper":"/paper/cost-effective-hyperparameter-optimization","title":"Cost-Effective Hyperparameter Optimization for Large Language Model Generation Inference","date":"2023-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevin666aa/flaml","path":"flaml/autogen/code_utils.py","file_url":"https://github.com/kevin666aa/flaml/blob/HEAD/flaml/autogen/code_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"0a954340369b7456","mcp_get_code":{"code_sha256":"0a954340369b7456"}}]}