{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/3","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":17,"rows_per_page":100,"rows":[201,300],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/2","next":"/task/code-generation/papers/4","papers":[{"url":"/paper/codeflowbench-a-multi-turn-iterative","slug":"codeflowbench-a-multi-turn-iterative","title":"CodeFlowBench: A Multi-turn, Iterative Benchmark for Complex Code Generation","date":"2025-04-30","arxiv_id":"2504.21751","repositories_listed":1,"syntology":null},{"url":"/paper/osvbench-benchmarking-llms-on-specification","slug":"osvbench-benchmarking-llms-on-specification","title":"OSVBench: Benchmarking LLMs on Specification Generation Tasks for Operating System Verification","date":"2025-04-29","arxiv_id":"2504.20964","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osvbench-benchmarking-llms-on-specification#ran","syntology_url":"https://syntology.ai/paper/2504.20964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20964"}},"official":{"repos":["lishangyu-hkust/osvbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reviving-any-subset-autoregressive-models","slug":"reviving-any-subset-autoregressive-models","title":"Reviving Any-Subset Autoregressive Models with Principled Parallel Sampling and Speculative Decoding","date":"2025-04-29","arxiv_id":"2504.20456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reviving-any-subset-autoregressive-models#ran","syntology_url":"https://syntology.ai/paper/2504.20456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20456"}},"official":{"repos":["gabeguo/any-order-speculative-decoding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/turing-machine-evaluation-for-large-language","slug":"turing-machine-evaluation-for-large-language","title":"Computational Reasoning of Large Language Models","date":"2025-04-29","arxiv_id":"2504.20771","repositories_listed":1,"syntology":null},{"url":"/paper/autop2c-an-llm-based-agent-framework-for-code","slug":"autop2c-an-llm-based-agent-framework-for-code","title":"AutoP2C: An LLM-Based Agent Framework for Code Repository Generation from Multimodal Content in Academic Papers","date":"2025-04-28","arxiv_id":"2504.20115","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autop2c-an-llm-based-agent-framework-for-code#ran","syntology_url":"https://syntology.ai/paper/2504.20115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20115"}},"official":{"repos":["shoushouyu/automated-paper-to-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codebc-a-more-secure-large-language-model-for","slug":"codebc-a-more-secure-large-language-model-for","title":"CodeBC: A More Secure Large Language Model for Smart Contract Code Generation in Blockchain","date":"2025-04-28","arxiv_id":"2504.21043","repositories_listed":1,"syntology":null},{"url":"/paper/chisellm-unleashing-the-power-of-reasoning","slug":"chisellm-unleashing-the-power-of-reasoning","title":"ChiseLLM: Unleashing the Power of Reasoning LLMs for Chisel Agile Hardware Development","date":"2025-04-27","arxiv_id":"2504.19144","repositories_listed":1,"syntology":null},{"url":"/paper/paper2code-automating-code-generation-from","slug":"paper2code-automating-code-generation-from","title":"Paper2Code: Automating Code Generation from Scientific Papers in Machine Learning","date":"2025-04-24","arxiv_id":"2504.17192","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paper2code-automating-code-generation-from#ran","syntology_url":"https://syntology.ai/paper/2504.17192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.17192"}},"official":{"repos":["going-doer/paper2code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-machine-generated-code-for-the","slug":"towards-machine-generated-code-for-the","title":"Towards Machine-Generated Code for the Resolution of User Intentions","date":"2025-04-24","arxiv_id":"2504.17531","repositories_listed":1,"syntology":null},{"url":"/paper/on-developers-self-declaration-of-ai","slug":"on-developers-self-declaration-of-ai","title":"On Developers' Self-Declaration of AI-Generated Code: An Analysis of Practices","date":"2025-04-23","arxiv_id":"2504.16485","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leetcodedataset-a-temporal-dataset-for-robust","slug":"leetcodedataset-a-temporal-dataset-for-robust","title":"LeetCodeDataset: A Temporal Dataset for Robust Evaluation and Efficient Training of Code LLMs","date":"2025-04-20","arxiv_id":"2504.14655","repositories_listed":1,"syntology":null},{"url":"/paper/reasoningv-efficient-verilog-code-generation","slug":"reasoningv-efficient-verilog-code-generation","title":"ReasoningV: Efficient Verilog Code Generation with Adaptive Hybrid Reasoning Model","date":"2025-04-20","arxiv_id":"2504.14560","repositories_listed":1,"syntology":null},{"url":"/paper/towards-optimal-circuit-generation-multi","slug":"towards-optimal-circuit-generation-multi","title":"Towards Optimal Circuit Generation: Multi-Agent Collaboration Meets Collective Intelligence","date":"2025-04-20","arxiv_id":"2504.14625","repositories_listed":1,"syntology":null},{"url":"/paper/chinese-vicuna-a-chinese-instruction","slug":"chinese-vicuna-a-chinese-instruction","title":"Chinese-Vicuna: A Chinese Instruction-following Llama-based Model","date":"2025-04-17","arxiv_id":"2504.12737","repositories_listed":1,"syntology":null},{"url":"/paper/robotwin-dual-arm-robot-benchmark-with-1","slug":"robotwin-dual-arm-robot-benchmark-with-1","title":"RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins","date":"2025-04-17","arxiv_id":"2504.13059","repositories_listed":1,"syntology":null},{"url":"/paper/a-dual-space-framework-for-general-knowledge","slug":"a-dual-space-framework-for-general-knowledge","title":"A Dual-Space Framework for General Knowledge Distillation of Large Language Models","date":"2025-04-15","arxiv_id":"2504.11426","repositories_listed":1,"syntology":null},{"url":"/paper/lori-reducing-cross-task-interference-in","slug":"lori-reducing-cross-task-interference-in","title":"LoRI: Reducing Cross-Task Interference in Multi-Task Low-Rank Adaptation","date":"2025-04-10","arxiv_id":"2504.07448","repositories_listed":1,"syntology":null},{"url":"/paper/do-llm-evaluators-prefer-themselves-for-a","slug":"do-llm-evaluators-prefer-themselves-for-a","title":"Do LLM Evaluators Prefer Themselves for a Reason?","date":"2025-04-04","arxiv_id":"2504.03846","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-chart-to-code-generation-in","slug":"enhancing-chart-to-code-generation-in","title":"Enhancing Chart-to-Code Generation in Multimodal Large Language Models via Iterative Dual Preference Learning","date":"2025-04-03","arxiv_id":"2504.02906","repositories_listed":1,"syntology":null},{"url":"/paper/maintaincoder-maintainable-code-generation","slug":"maintaincoder-maintainable-code-generation","title":"MaintainCoder: Maintainable Code Generation Under Dynamic Requirements","date":"2025-03-31","arxiv_id":"2503.24260","repositories_listed":1,"syntology":null},{"url":"/paper/turtle-a-unified-evaluation-of-llms-for-rtl","slug":"turtle-a-unified-evaluation-of-llms-for-rtl","title":"TuRTLe: A Unified Evaluation of LLMs for RTL Generation","date":"2025-03-31","arxiv_id":"2504.01986","repositories_listed":1,"syntology":null},{"url":"/paper/why-stop-at-one-error-benchmarking-llms-as","slug":"why-stop-at-one-error-benchmarking-llms-as","title":"Why Stop at One Error? Benchmarking LLMs as Data Science Code Debuggers for Multi-Hop and Multi-Bug Errors","date":"2025-03-28","arxiv_id":"2503.22388","repositories_listed":1,"syntology":null},{"url":"/paper/obscuracoder-powering-efficient-code-lm-pre","slug":"obscuracoder-powering-efficient-code-lm-pre","title":"ObscuraCoder: Powering Efficient Code LM Pre-Training Via Obfuscation Grounding","date":"2025-03-27","arxiv_id":"2504.00019","repositories_listed":1,"syntology":null},{"url":"/paper/rustevo-2-an-evolving-benchmark-for-api","slug":"rustevo-2-an-evolving-benchmark-for-api","title":"RustEvo^2: An Evolving Benchmark for API Evolution in LLM-based Rust Code Generation","date":"2025-03-21","arxiv_id":"2503.16922","repositories_listed":1,"syntology":null},{"url":"/paper/bigo-bench-can-llms-generate-code-with","slug":"bigo-bench-can-llms-generate-code-with","title":"BigO(Bench) -- Can LLMs Generate Code with Controlled Time and Space Complexity?","date":"2025-03-19","arxiv_id":"2503.15242","repositories_listed":1,"syntology":null},{"url":"/paper/llm-hpc-benchmarking-deepseek-s-performance","slug":"llm-hpc-benchmarking-deepseek-s-performance","title":"LLM & HPC:Benchmarking DeepSeek's Performance in High-Performance Computing Tasks","date":"2025-03-15","arxiv_id":"2504.03665","repositories_listed":1,"syntology":null},{"url":"/paper/unified-modeling-language-code-generation","slug":"unified-modeling-language-code-generation","title":"Unified Modeling Language Code Generation from Diagram Images Using Multimodal Large Language Models","date":"2025-03-15","arxiv_id":"2503.12293","repositories_listed":1,"syntology":null},{"url":"/paper/asma-tune-unlocking-llms-assembly-code","slug":"asma-tune-unlocking-llms-assembly-code","title":"ASMA-Tune: Unlocking LLMs' Assembly Code Comprehension via Structural-Semantic Instruction Tuning","date":"2025-03-14","arxiv_id":"2503.11617","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-high-quality-code-generation-in","slug":"enhancing-high-quality-code-generation-in","title":"Enhancing High-Quality Code Generation in Large Language Models with Comparative Prefix-Tuning","date":"2025-03-12","arxiv_id":"2503.09020","repositories_listed":1,"syntology":null},{"url":"/paper/distillm-2-a-contrastive-approach-boosts-the","slug":"distillm-2-a-contrastive-approach-boosts-the","title":"DistiLLM-2: A Contrastive Approach Boosts the Distillation of LLMs","date":"2025-03-10","arxiv_id":"2503.07067","repositories_listed":1,"syntology":null},{"url":"/paper/projecteval-a-benchmark-for-programming","slug":"projecteval-a-benchmark-for-programming","title":"ProjectEval: A Benchmark for Programming Agents Automated Evaluation on Project-Level Code Generation","date":"2025-03-10","arxiv_id":"2503.07010","repositories_listed":1,"syntology":null},{"url":"/paper/repost-scalable-repository-level-coding","slug":"repost-scalable-repository-level-coding","title":"RepoST: Scalable Repository-Level Coding Environment Construction with Sandbox Testing","date":"2025-03-10","arxiv_id":"2503.07358","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/repost-scalable-repository-level-coding#ran","syntology_url":"https://syntology.ai/paper/2503.07358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07358"}},"official":{"repos":["yiqingxyq/RepoST"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/dependeval-benchmarking-llms-for-repository","slug":"dependeval-benchmarking-llms-for-repository","title":"DependEval: Benchmarking LLMs for Repository Dependency Understanding","date":"2025-03-09","arxiv_id":"2503.06689","repositories_listed":1,"syntology":null},{"url":"/paper/2503-00686","slug":"2503-00686","title":"GPIoT: Tailoring Small Language Models for IoT Program Synthesis and Development","date":"2025-03-02","arxiv_id":"2503.00686","repositories_listed":1,"syntology":null},{"url":"/paper/multi-turn-code-generation-through-single","slug":"multi-turn-code-generation-through-single","title":"Multi-Turn Code Generation Through Single-Step Rewards","date":"2025-02-27","arxiv_id":"2502.20380","repositories_listed":1,"syntology":null},{"url":"/paper/codeif-benchmarking-the-instruction-following","slug":"codeif-benchmarking-the-instruction-following","title":"CodeIF: Benchmarking the Instruction-Following Capabilities of Large Language Models for Code Generation","date":"2025-02-26","arxiv_id":"2502.19166","repositories_listed":1,"syntology":null},{"url":"/paper/indiceval-xl-bridging-linguistic-diversity-in","slug":"indiceval-xl-bridging-linguistic-diversity-in","title":"IndicEval-XL: Bridging Linguistic Diversity in Code Generation Across Indic Languages","date":"2025-02-26","arxiv_id":"2502.19067","repositories_listed":1,"syntology":null},{"url":"/paper/nexus-a-lightweight-and-scalable-multi-agent","slug":"nexus-a-lightweight-and-scalable-multi-agent","title":"Nexus: A Lightweight and Scalable Multi-Agent Framework for Complex Tasks Automation","date":"2025-02-26","arxiv_id":"2502.19091","repositories_listed":1,"syntology":null},{"url":"/paper/program-synthesis-dialog-agents-for","slug":"program-synthesis-dialog-agents-for","title":"Program Synthesis Dialog Agents for Interactive Decision-Making","date":"2025-02-26","arxiv_id":"2502.19610","repositories_listed":1,"syntology":null},{"url":"/paper/an-analyst-inspector-framework-for-evaluating","slug":"an-analyst-inspector-framework-for-evaluating","title":"An Analyst-Inspector Framework for Evaluating Reproducibility of LLMs in Data Science","date":"2025-02-23","arxiv_id":"2502.16395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-analyst-inspector-framework-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.16395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16395"}},"official":{"repos":["qunhualilab/llm-ds-reproducibility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codecriticbench-a-holistic-code-critique","slug":"codecriticbench-a-holistic-code-critique","title":"CodeCriticBench: A Holistic Code Critique Benchmark for Large Language Models","date":"2025-02-23","arxiv_id":"2502.16614","repositories_listed":1,"syntology":null},{"url":"/paper/gate-graph-based-adaptive-tool-evolution","slug":"gate-graph-based-adaptive-tool-evolution","title":"GATE: Graph-based Adaptive Tool Evolution Across Diverse Tasks","date":"2025-02-20","arxiv_id":"2502.14848","repositories_listed":1,"syntology":null},{"url":"/paper/i-mcts-enhancing-agentic-automl-via","slug":"i-mcts-enhancing-agentic-automl-via","title":"I-MCTS: Enhancing Agentic AutoML via Introspective Monte Carlo Tree Search","date":"2025-02-20","arxiv_id":"2502.14693","repositories_listed":1,"syntology":null},{"url":"/paper/s-test-time-scaling-for-code-generation","slug":"s-test-time-scaling-for-code-generation","title":"S*: Test Time Scaling for Code Generation","date":"2025-02-20","arxiv_id":"2502.14382","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-test-time-scaling-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.14382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14382"}},"official":{"repos":["novasky-ai/skythought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tritonbench-benchmarking-large-language-model","slug":"tritonbench-benchmarking-large-language-model","title":"TritonBench: Benchmarking Large Language Model Capabilities for Generating Triton Operators","date":"2025-02-20","arxiv_id":"2502.14752","repositories_listed":1,"syntology":null},{"url":"/paper/adaptivestep-automatically-dividing-reasoning","slug":"adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","arxiv_id":"2502.13943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptivestep-automatically-dividing-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.13943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13943"}},"official":{"repos":["lux0926/asprm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/datascibench-an-llm-agent-benchmark-for-data","slug":"datascibench-an-llm-agent-benchmark-for-data","title":"DataSciBench: An LLM Agent Benchmark for Data Science","date":"2025-02-19","arxiv_id":"2502.13897","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-agents-to-overcome-ambiguity-in","slug":"interactive-agents-to-overcome-ambiguity-in","title":"Interactive Agents to Overcome Ambiguity in Software Engineering","date":"2025-02-18","arxiv_id":"2502.13069","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-agents-to-overcome-ambiguity-in#ran","syntology_url":"https://syntology.ai/paper/2502.13069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13069"}},"official":{"repos":["sani903/interactivesweagents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/performance-evaluation-of-large-language","slug":"performance-evaluation-of-large-language","title":"Performance Evaluation of Large Language Models in Statistical Programming","date":"2025-02-18","arxiv_id":"2502.13117","repositories_listed":1,"syntology":null},{"url":"/paper/training-turn-by-turn-verifiers-for-dialogue","slug":"training-turn-by-turn-verifiers-for-dialogue","title":"Training Turn-by-Turn Verifiers for Dialogue Tutoring Agents: The Curious Case of LLMs as Your Coding Tutors","date":"2025-02-18","arxiv_id":"2502.13311","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-turn-by-turn-verifiers-for-dialogue#ran","syntology_url":"https://syntology.ai/paper/2502.13311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13311"}},"official":{"repos":["iwangjian/Coding-Tutor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unigencoder-merging-seq2seq-and-seq2tree","slug":"unigencoder-merging-seq2seq-and-seq2tree","title":"UniGenCoder: Merging Seq2Seq and Seq2Tree Paradigms for Unified Code Generation","date":"2025-02-18","arxiv_id":"2502.12490","repositories_listed":1,"syntology":null},{"url":"/paper/code-vision-evaluating-multimodal-llms-logic","slug":"code-vision-evaluating-multimodal-llms-logic","title":"Code-Vision: Evaluating Multimodal LLMs Logic Understanding and Code Generation Capabilities","date":"2025-02-17","arxiv_id":"2502.11829","repositories_listed":1,"syntology":null},{"url":"/paper/gift-gibbs-fine-tuning-for-code-generation","slug":"gift-gibbs-fine-tuning-for-code-generation","title":"GiFT: Gibbs Fine-Tuning for Code Generation","date":"2025-02-17","arxiv_id":"2502.11466","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/gift-gibbs-fine-tuning-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.11466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11466"}},"official":{"repos":["alex-haochenli/gift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toolcoder-a-systematic-code-empowered-tool","slug":"toolcoder-a-systematic-code-empowered-tool","title":"ToolCoder: A Systematic Code-Empowered Tool Learning Framework for Large Language Models","date":"2025-02-17","arxiv_id":"2502.11404","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-the-impact-of-chain-of-thought","slug":"uncovering-the-impact-of-chain-of-thought","title":"Uncovering the Impact of Chain-of-Thought Reasoning for Direct Preference Optimization: Lessons from Text-to-SQL","date":"2025-02-17","arxiv_id":"2502.11656","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-cross-tokenizer-knowledge","slug":"enhancing-cross-tokenizer-knowledge","title":"Enhancing Cross-Tokenizer Knowledge Distillation with Contextual Dynamical Mapping","date":"2025-02-16","arxiv_id":"2502.11104","repositories_listed":1,"syntology":null},{"url":"/paper/surge-on-the-potential-of-large-language","slug":"surge-on-the-potential-of-large-language","title":"SURGE: On the Potential of Large Language Models as General-Purpose Surrogate Code Executors","date":"2025-02-16","arxiv_id":"2502.11167","repositories_listed":1,"syntology":null},{"url":"/paper/cocoevo-co-evolution-of-programs-and-test","slug":"cocoevo-co-evolution-of-programs-and-test","title":"CoCoEvo: Co-Evolution of Programs and Test Cases to Enhance Code Generation","date":"2025-02-15","arxiv_id":"2502.10802","repositories_listed":1,"syntology":null},{"url":"/paper/benchmax-a-comprehensive-multilingual","slug":"benchmax-a-comprehensive-multilingual","title":"BenchMAX: A Comprehensive Multilingual Evaluation Suite for Large Language Models","date":"2025-02-11","arxiv_id":"2502.07346","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-llm-generated-code-and-requirements","slug":"bridging-llm-generated-code-and-requirements","title":"Bridging LLM-Generated Code and Requirements: Reverse Generation technique and SBC Metric for Developer Insights","date":"2025-02-11","arxiv_id":"2502.07835","repositories_listed":1,"syntology":null},{"url":"/paper/codei-o-condensing-reasoning-patterns-via","slug":"codei-o-condensing-reasoning-patterns-via","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","date":"2025-02-11","arxiv_id":"2502.07316","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/codei-o-condensing-reasoning-patterns-via#ran","syntology_url":"https://syntology.ai/paper/2502.07316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07316"}},"official":{"repos":["hkust-nlp/codeio"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codesim-multi-agent-code-generation-and-1","slug":"codesim-multi-agent-code-generation-and-1","title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","date":"2025-02-08","arxiv_id":"2502.05664","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codesim-multi-agent-code-generation-and-1#ran","syntology_url":"https://syntology.ai/paper/2502.05664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05664"}},"official":null}},{"url":"/paper/codescm-causal-analysis-for-multi-modal-code","slug":"codescm-causal-analysis-for-multi-modal-code","title":"CodeSCM: Causal Analysis for Multi-Modal Code Generation","date":"2025-02-07","arxiv_id":"2502.05150","repositories_listed":1,"syntology":null},{"url":"/paper/nvagent-automated-data-visualization-from","slug":"nvagent-automated-data-visualization-from","title":"nvAgent: Automated Data Visualization from Natural Language via Collaborative Agent Workflow","date":"2025-02-07","arxiv_id":"2502.05036","repositories_listed":1,"syntology":null},{"url":"/paper/codesteer-symbolic-augmented-language-models","slug":"codesteer-symbolic-augmented-language-models","title":"CodeSteer: Symbolic-Augmented Language Models via Code/Text Guidance","date":"2025-02-04","arxiv_id":"2502.04350","repositories_listed":1,"syntology":null},{"url":"/paper/longdpo-unlock-better-long-form-generation","slug":"longdpo-unlock-better-long-form-generation","title":"LongDPO: Unlock Better Long-form Generation Abilities for LLMs via Critique-augmented Stepwise Information","date":"2025-02-04","arxiv_id":"2502.02095","repositories_listed":1,"syntology":null},{"url":"/paper/reusing-embeddings-reproducible-reward-model","slug":"reusing-embeddings-reproducible-reward-model","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","date":"2025-02-04","arxiv_id":"2502.04357","repositories_listed":1,"syntology":null},{"url":"/paper/the-elicitation-game-evaluating-capability","slug":"the-elicitation-game-evaluating-capability","title":"The Elicitation Game: Evaluating Capability Elicitation Techniques","date":"2025-02-04","arxiv_id":"2502.02180","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-elicitation-game-evaluating-capability#ran","syntology_url":"https://syntology.ai/paper/2502.02180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02180"}},"official":null}},{"url":"/paper/c-code-generation-considered-unnecessary-go","slug":"c-code-generation-considered-unnecessary-go","title":"C codegen considered unnecessary: go directly to binary, do not pass C. Compilation of Julia code for deployment in model-based engineering","date":"2025-02-03","arxiv_id":"2502.01128","repositories_listed":1,"syntology":null},{"url":"/paper/cogito-ergo-sum-a-neurobiologically-inspired","slug":"cogito-ergo-sum-a-neurobiologically-inspired","title":"Cogito, ergo sum: A Neurobiologically-Inspired Cognition-Memory-Growth System for Code Generation","date":"2025-01-30","arxiv_id":"2501.18653","repositories_listed":1,"syntology":null},{"url":"/paper/o3-mini-vs-deepseek-r1-which-one-is-safer","slug":"o3-mini-vs-deepseek-r1-which-one-is-safer","title":"o3-mini vs DeepSeek-R1: Which One is Safer?","date":"2025-01-30","arxiv_id":"2501.18438","repositories_listed":1,"syntology":null},{"url":"/paper/towards-making-flowchart-images-machine-1","slug":"towards-making-flowchart-images-machine-1","title":"Towards Making Flowchart Images Machine Interpretable","date":"2025-01-29","arxiv_id":"2501.17441","repositories_listed":1,"syntology":null},{"url":"/paper/using-code-generation-to-solve-open-instances","slug":"using-code-generation-to-solve-open-instances","title":"Using Code Generation to Solve Open Instances of Combinatorial Design Problems","date":"2025-01-29","arxiv_id":"2501.17725","repositories_listed":1,"syntology":null},{"url":"/paper/coconut-structural-code-understanding-does","slug":"coconut-structural-code-understanding-does","title":"CoCoNUT: Structural Code Understanding does not fall out of a tree","date":"2025-01-27","arxiv_id":"2501.16456","repositories_listed":1,"syntology":null},{"url":"/paper/correctness-assessment-of-code-generated-by","slug":"correctness-assessment-of-code-generated-by","title":"Correctness Assessment of Code Generated by Large Language Models Using Internal Representations","date":"2025-01-22","arxiv_id":"2501.12934","repositories_listed":1,"syntology":null},{"url":"/paper/chaoseater-fully-automating-chaos-engineering","slug":"chaoseater-fully-automating-chaos-engineering","title":"ChaosEater: Fully Automating Chaos Engineering with Large Language Models","date":"2025-01-19","arxiv_id":"2501.11107","repositories_listed":1,"syntology":null},{"url":"/paper/green-code-optimizing-energy-efficiency-in","slug":"green-code-optimizing-energy-efficiency-in","title":"GREEN-CODE: Learning to Optimize Energy Efficiency in LLM-based Code Generation","date":"2025-01-19","arxiv_id":"2501.11006","repositories_listed":1,"syntology":null},{"url":"/paper/cweval-outcome-driven-evaluation-on","slug":"cweval-outcome-driven-evaluation-on","title":"CWEval: Outcome-driven Evaluation on Functionality and Security of LLM Code Generation","date":"2025-01-14","arxiv_id":"2501.08200","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cweval-outcome-driven-evaluation-on#ran","syntology_url":"https://syntology.ai/paper/2501.08200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08200"}},"official":{"repos":["co1lin/cweval"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optichat-bridging-optimization-models-and","slug":"optichat-bridging-optimization-models-and","title":"OptiChat: Bridging Optimization Models and Practitioners with Large Language Models","date":"2025-01-14","arxiv_id":"2501.08406","repositories_listed":1,"syntology":null},{"url":"/paper/chartcoder-advancing-multimodal-large","slug":"chartcoder-advancing-multimodal-large","title":"ChartCoder: Advancing Multimodal Large Language Model for Chart-to-Code Generation","date":"2025-01-11","arxiv_id":"2501.06598","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/chartcoder-advancing-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2501.06598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.06598"}},"official":{"repos":["thunlp/ChartCoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/how-to-select-pre-trained-code-models-for","slug":"how-to-select-pre-trained-code-models-for","title":"How to Select Pre-Trained Code Models for Reuse? A Learning Perspective","date":"2025-01-07","arxiv_id":"2501.03783","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-captioning-benchmark-dataset-and","slug":"semantic-captioning-benchmark-dataset-and","title":"Semantic Captioning: Benchmark Dataset and Graph-Aware Few-Shot In-Context Learning for SQL2Text","date":"2025-01-06","arxiv_id":"2501.03166","repositories_listed":1,"syntology":null},{"url":"/paper/layer-level-self-exposure-and-patch","slug":"layer-level-self-exposure-and-patch","title":"Layer-Level Self-Exposure and Patch: Affirmative Token Mitigation for Jailbreak Attack Defense","date":"2025-01-05","arxiv_id":"2501.02629","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-the-code-generation","slug":"evaluation-of-the-code-generation","title":"Evaluation of the Code Generation Capabilities of ChatGPT 4: A Comparative Analysis in 19 Programming Languages","date":"2025-01-04","arxiv_id":"2501.02338","repositories_listed":1,"syntology":null},{"url":"/paper/effective-llm-driven-code-generation-with","slug":"effective-llm-driven-code-generation-with","title":"Effective LLM-Driven Code Generation with Pythoness","date":"2025-01-03","arxiv_id":"2501.02138","repositories_listed":1,"syntology":null},{"url":"/paper/automated-self-refinement-and-self-correction","slug":"automated-self-refinement-and-self-correction","title":"Automated Self-Refinement and Self-Correction for LLM-based Product Attribute Value Extraction","date":"2025-01-02","arxiv_id":"2501.01237","repositories_listed":1,"syntology":null},{"url":"/paper/efficiently-serving-llm-reasoning-programs","slug":"efficiently-serving-llm-reasoning-programs","title":"Efficiently Serving LLM Reasoning Programs with Certaindex","date":"2024-12-30","arxiv_id":"2412.20993","repositories_listed":1,"syntology":null},{"url":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large","slug":"humaneval-pro-and-mbpp-pro-evaluating-large","title":"HumanEval Pro and MBPP Pro: Evaluating Large Language Models on Self-invoking Code Generation","date":"2024-12-30","arxiv_id":"2412.21199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2412.21199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21199"}},"official":{"repos":["CodeEval-Pro/CodeEval-Pro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-intelligent-and-secure-cloud-large","slug":"toward-intelligent-and-secure-cloud-large","title":"Toward Intelligent and Secure Cloud: Large Language Model Empowered Proactive Defense","date":"2024-12-30","arxiv_id":"2412.21051","repositories_listed":1,"syntology":null},{"url":"/paper/the-impact-of-prompt-programming-on-function","slug":"the-impact-of-prompt-programming-on-function","title":"The Impact of Prompt Programming on Function-Level Code Generation","date":"2024-12-29","arxiv_id":"2412.20545","repositories_listed":1,"syntology":null},{"url":"/paper/autodroid-v2-boosting-slm-based-gui-agents","slug":"autodroid-v2-boosting-slm-based-gui-agents","title":"AutoDroid-V2: Boosting SLM-based GUI Agents via Code Generation","date":"2024-12-24","arxiv_id":"2412.18116","repositories_listed":1,"syntology":null},{"url":"/paper/outcome-refining-process-supervision-for-code","slug":"outcome-refining-process-supervision-for-code","title":"Reasoning Through Execution: Unifying Process and Outcome Rewards for Code Generation","date":"2024-12-19","arxiv_id":"2412.15118","repositories_listed":1,"syntology":null},{"url":"/paper/seeker-towards-exception-safety-code","slug":"seeker-towards-exception-safety-code","title":"Seeker: Towards Exception Safety Code Generation with Intermediate Language Agents Framework","date":"2024-12-16","arxiv_id":"2412.11713","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeker-towards-exception-safety-code#ran","syntology_url":"https://syntology.ai/paper/2412.11713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11713"}},"official":{"repos":["XMZhangAI/Seeker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chainstream-an-llm-based-framework-for","slug":"chainstream-an-llm-based-framework-for","title":"ChainStream: An LLM-based Framework for Unified Synthetic Sensing","date":"2024-12-13","arxiv_id":"2412.15240","repositories_listed":1,"syntology":null},{"url":"/paper/mage-a-multi-agent-engine-for-automated-rtl","slug":"mage-a-multi-agent-engine-for-automated-rtl","title":"MAGE: A Multi-Agent Engine for Automated RTL Code Generation","date":"2024-12-10","arxiv_id":"2412.07822","repositories_listed":1,"syntology":null},{"url":"/paper/copyright-protected-language-generation-via","slug":"copyright-protected-language-generation-via","title":"Copyright-Protected Language Generation via Adaptive Model Fusion","date":"2024-12-09","arxiv_id":"2412.06619","repositories_listed":1,"syntology":null},{"url":"/paper/towards-rich-emotions-in-3d-avatars-a-text-to","slug":"towards-rich-emotions-in-3d-avatars-a-text-to","title":"Towards Rich Emotions in 3D Avatars: A Text-to-3D Avatar Generation Benchmark","date":"2024-12-03","arxiv_id":"2412.02508","repositories_listed":1,"syntology":null},{"url":"/paper/commit0-library-generation-from-scratch","slug":"commit0-library-generation-from-scratch","title":"Commit0: Library Generation from Scratch","date":"2024-12-02","arxiv_id":"2412.01769","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commit0-library-generation-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2412.01769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01769"}},"official":{"repos":["commit-0/commit0"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"41be7b6427cb46fadc5406a9a5a4364eb165660068b09c609ef3a1bb99e3f57a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}