{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/5","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":17,"rows_per_page":100,"rows":[401,500],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/4","next":"/task/code-generation/papers/6","papers":[{"url":"/paper/origen-enhancing-rtl-code-generation-with","slug":"origen-enhancing-rtl-code-generation-with","title":"OriGen:Enhancing RTL Code Generation with Code-to-Code Augmentation and Self-Reflection","date":"2024-07-23","arxiv_id":"2407.16237","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/origen-enhancing-rtl-code-generation-with#ran","syntology_url":"https://syntology.ai/paper/2407.16237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16237"}},"official":{"repos":["pku-liang/origen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-for-verilog-generation","slug":"large-language-model-for-verilog-generation","title":"Large Language Model for Verilog Generation with Code-Structure-Guided Reinforcement Learning","date":"2024-07-21","arxiv_id":"2407.18271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-model-for-verilog-generation#ran","syntology_url":"https://syntology.ai/paper/2407.18271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18271"}},"official":{"repos":["CatIIIIIIII/veriseek"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-impact-of-pursuing-modularity","slug":"revisiting-the-impact-of-pursuing-modularity","title":"Revisiting the Impact of Pursuing Modularity for Code Generation","date":"2024-07-16","arxiv_id":"2407.11406","repositories_listed":1,"syntology":null},{"url":"/paper/spider2-v-how-far-are-multimodal-agents-from","slug":"spider2-v-how-far-are-multimodal-agents-from","title":"Spider2-V: How Far Are Multimodal Agents From Automating Data Science and Engineering Workflows?","date":"2024-07-15","arxiv_id":"2407.10956","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spider2-v-how-far-are-multimodal-agents-from#ran","syntology_url":"https://syntology.ai/paper/2407.10956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10956"}},"official":{"repos":["xlang-ai/spider2-v"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-language-model-creativity-a-case#ran","syntology_url":"https://syntology.ai/paper/2407.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09007"}},"official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inversecoder-unleashing-the-power-of#ran","syntology_url":"https://syntology.ai/paper/2407.05700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05700"}},"official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/oran-bench-13k-an-open-source-benchmark-for","slug":"oran-bench-13k-an-open-source-benchmark-for","title":"ORAN-Bench-13K: An Open Source Benchmark for Assessing LLMs in Open Radio Access Networks","date":"2024-07-08","arxiv_id":"2407.06245","repositories_listed":1,"syntology":null},{"url":"/paper/data-overfitting-for-on-device-super","slug":"data-overfitting-for-on-device-super","title":"Data Overfitting for On-Device Super-Resolution with Dynamic Algorithm and Compiler Co-Design","date":"2024-07-03","arxiv_id":"2407.02813","repositories_listed":1,"syntology":null},{"url":"/paper/theoremllama-transforming-general-purpose","slug":"theoremllama-transforming-general-purpose","title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","date":"2024-07-03","arxiv_id":"2407.03203","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/theoremllama-transforming-general-purpose#ran","syntology_url":"https://syntology.ai/paper/2407.03203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03203"}},"official":{"repos":["RickySkywalker/TheoremLlama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-monoculture-in-large-language","slug":"generative-monoculture-in-large-language","title":"Generative Monoculture in Large Language Models","date":"2024-07-02","arxiv_id":"2407.02209","repositories_listed":1,"syntology":null},{"url":"/paper/integrate-the-essence-and-eliminate-the-dross","slug":"integrate-the-essence-and-eliminate-the-dross","title":"Integrate the Essence and Eliminate the Dross: Fine-Grained Self-Consistency for Free-Form Language Generation","date":"2024-07-02","arxiv_id":"2407.02056","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/integrate-the-essence-and-eliminate-the-dross#ran","syntology_url":"https://syntology.ai/paper/2407.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02056"}},"official":{"repos":["WangXinglin/FSC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/discoverybench-towards-data-driven-discovery","slug":"discoverybench-towards-data-driven-discovery","title":"DiscoveryBench: Towards Data-Driven Discovery with Large Language Models","date":"2024-07-01","arxiv_id":"2407.01725","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discoverybench-towards-data-driven-discovery#ran","syntology_url":"https://syntology.ai/paper/2407.01725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01725"}},"official":{"repos":["allenai/discoverybench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/web2code-a-large-scale-webpage-to-code","slug":"web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","arxiv_id":"2406.20098","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web2code-a-large-scale-webpage-to-code#ran","syntology_url":"https://syntology.ai/paper/2406.20098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20098"}},"official":{"repos":["mbzuai-llm/web2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mpcoder-multi-user-personalized-code","slug":"mpcoder-multi-user-personalized-code","title":"MPCODER: Multi-user Personalized Code Generator with Explicit and Implicit Style Representation Learning","date":"2024-06-25","arxiv_id":"2406.17255","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mpcoder-multi-user-personalized-code#ran","syntology_url":"https://syntology.ai/paper/2406.17255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17255"}},"official":{"repos":["455849940/MPCoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beear-embedding-based-adversarial-removal-of","slug":"beear-embedding-based-adversarial-removal-of","title":"BEEAR: Embedding-based Adversarial Removal of Safety Backdoors in Instruction-tuned Language Models","date":"2024-06-24","arxiv_id":"2406.17092","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beear-embedding-based-adversarial-removal-of#ran","syntology_url":"https://syntology.ai/paper/2406.17092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17092"}},"official":{"repos":["reds-lab/beear"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/res-q-evaluating-code-editing-large-language","slug":"res-q-evaluating-code-editing-large-language","title":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","date":"2024-06-24","arxiv_id":"2406.16801","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/res-q-evaluating-code-editing-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16801"}},"official":{"repos":["qurrent-ai/res-q"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/indict-code-generation-with-internal","slug":"indict-code-generation-with-internal","title":"INDICT: Code Generation with Internal Dialogues of Critiques for Both Security and Helpfulness","date":"2024-06-23","arxiv_id":"2407.02518","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/indict-code-generation-with-internal#ran","syntology_url":"https://syntology.ai/paper/2407.02518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02518"}},"official":{"repos":["SalesforceAIResearch/indict_code_gen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genotex-a-benchmark-for-evaluating-llm-based","slug":"genotex-a-benchmark-for-evaluating-llm-based","title":"GenoTEX: An LLM Agent Benchmark for Automated Gene Expression Data Analysis","date":"2024-06-21","arxiv_id":"2406.15341","repositories_listed":1,"syntology":null},{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coderag-bench-can-retrieval-augment-code","slug":"coderag-bench-can-retrieval-augment-code","title":"CodeRAG-Bench: Can Retrieval Augment Code Generation?","date":"2024-06-20","arxiv_id":"2406.14497","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderag-bench-can-retrieval-augment-code#ran","syntology_url":"https://syntology.ai/paper/2406.14497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14497"}},"official":{"repos":["code-rag-bench/code-rag-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-agents-are-state-of-the-art-software","slug":"code-agents-are-state-of-the-art-software","title":"SWT-Bench: Testing and Validating Real-World Bug-Fixes with Code Agents","date":"2024-06-18","arxiv_id":"2406.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/code-agents-are-state-of-the-art-software#ran","syntology_url":"https://syntology.ai/paper/2406.12952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12952"}},"official":{"repos":["logic-star-ai/swt-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gw-moe-resolving-uncertainty-in-moe-router","slug":"gw-moe-resolving-uncertainty-in-moe-router","title":"GW-MoE: Resolving Uncertainty in MoE Router with Global Workspace Theory","date":"2024-06-18","arxiv_id":"2406.12375","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/cosqa-enhancing-code-search-dataset-with","slug":"cosqa-enhancing-code-search-dataset-with","title":"CoSQA+: Pioneering the Multi-Choice Code Search Benchmark with Test-Driven Agents","date":"2024-06-17","arxiv_id":"2406.11589","repositories_listed":1,"syntology":null},{"url":"/paper/github-copilot-the-perfect-code-compleeter","slug":"github-copilot-the-perfect-code-compleeter","title":"GitHub Copilot: the perfect Code compLeeter?","date":"2024-06-17","arxiv_id":"2406.11326","repositories_listed":1,"syntology":null},{"url":"/paper/long-code-arena-a-set-of-benchmarks-for-long","slug":"long-code-arena-a-set-of-benchmarks-for-long","title":"Long Code Arena: a Set of Benchmarks for Long-Context Code Models","date":"2024-06-17","arxiv_id":"2406.11612","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-code-arena-a-set-of-benchmarks-for-long#ran","syntology_url":"https://syntology.ai/paper/2406.11612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11612"}},"official":{"repos":["jetbrains-research/lca-baselines"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/repoexec-evaluate-code-generation-with-a","slug":"repoexec-evaluate-code-generation-with-a","title":"On the Impacts of Contexts on Repository-Level Code Generation","date":"2024-06-17","arxiv_id":"2406.11927","repositories_listed":1,"syntology":null},{"url":"/paper/agilecoder-dynamic-collaborative-agents-for","slug":"agilecoder-dynamic-collaborative-agents-for","title":"AgileCoder: Dynamic Collaborative Agents for Software Development based on Agile Methodology","date":"2024-06-16","arxiv_id":"2406.11912","repositories_listed":1,"syntology":null},{"url":"/paper/chartmimic-evaluating-lmm-s-cross-modal","slug":"chartmimic-evaluating-lmm-s-cross-modal","title":"ChartMimic: Evaluating LMM's Cross-Modal Reasoning Capability via Chart-to-Code Generation","date":"2024-06-14","arxiv_id":"2406.09961","repositories_listed":1,"syntology":null},{"url":"/paper/cs-bench-a-comprehensive-benchmark-for-large","slug":"cs-bench-a-comprehensive-benchmark-for-large","title":"CS-Bench: A Comprehensive Benchmark for Large Language Models towards Computer Science Mastery","date":"2024-06-12","arxiv_id":"2406.08587","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cs-bench-a-comprehensive-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.08587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08587"}},"official":{"repos":["csbench/csbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-have-a-package-for-you-a-comprehensive","slug":"we-have-a-package-for-you-a-comprehensive","title":"We Have a Package for You! A Comprehensive Analysis of Package Hallucinations by Code Generating LLMs","date":"2024-06-12","arxiv_id":"2406.10279","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-reinforced-planning-with-large","slug":"entropy-reinforced-planning-with-large","title":"Entropy-Reinforced Planning with Large Language Models for Drug Discovery","date":"2024-06-11","arxiv_id":"2406.07025","repositories_listed":1,"syntology":null},{"url":"/paper/versicode-towards-version-controllable-code","slug":"versicode-towards-version-controllable-code","title":"VersiCode: Towards Version-controllable Code Generation","date":"2024-06-11","arxiv_id":"2406.07411","repositories_listed":1,"syntology":null},{"url":"/paper/can-ai-beat-undergraduates-in-entry-level","slug":"can-ai-beat-undergraduates-in-entry-level","title":"JavaBench: A Benchmark of Object-Oriented Code Generation for Evaluating Large Language Models","date":"2024-06-10","arxiv_id":"2406.12902","repositories_listed":1,"syntology":null},{"url":"/paper/sinklora-enhanced-efficiency-and-chat","slug":"sinklora-enhanced-efficiency-and-chat","title":"SinkLoRA: Enhanced Efficiency and Chat Capabilities for Long-Context Large Language Models","date":"2024-06-09","arxiv_id":"2406.05678","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sinklora-enhanced-efficiency-and-chat#ran","syntology_url":"https://syntology.ai/paper/2406.05678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05678"}},"official":{"repos":["dexter-gt-86/sinklora"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/logicode-an-llm-driven-framework-for-logical","slug":"logicode-an-llm-driven-framework-for-logical","title":"LogiCode: an LLM-Driven Framework for Logical Anomaly Detection","date":"2024-06-07","arxiv_id":"2406.04687","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"0 ran · 5 unverified","sample_list":"/paper/logicode-an-llm-driven-framework-for-logical#ran","syntology_url":"https://syntology.ai/paper/2406.04687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04687"}},"official":{"repos":["22strongestme/LOCO-Annotations"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/online-joint-fine-tuning-of-multi-agent-flows","slug":"online-joint-fine-tuning-of-multi-agent-flows","title":"Online Joint Fine-tuning of Multi-Agent Flows","date":"2024-06-06","arxiv_id":"2406.04516","repositories_listed":1,"syntology":null},{"url":"/paper/reflection-reinforced-self-training-for","slug":"reflection-reinforced-self-training-for","title":"Re-ReST: Reflection-Reinforced Self-Training for Language Agents","date":"2024-06-03","arxiv_id":"2406.01495","repositories_listed":1,"syntology":null},{"url":"/paper/semcoder-training-code-language-models-with","slug":"semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","arxiv_id":"2406.01006","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/semcoder-training-code-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2406.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01006"}},"official":{"repos":["arise-lab/semcoder"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deveval-a-manually-annotated-code-generation","slug":"deveval-a-manually-annotated-code-generation","title":"DevEval: A Manually-Annotated Code Generation Benchmark Aligned with Real-World Code Repositories","date":"2024-05-30","arxiv_id":"2405.19856","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deveval-a-manually-annotated-code-generation#ran","syntology_url":"https://syntology.ai/paper/2405.19856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19856"}},"official":{"repos":["seketeam/deveval"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploiting-llm-quantization","slug":"exploiting-llm-quantization","title":"Exploiting LLM Quantization","date":"2024-05-28","arxiv_id":"2405.18137","repositories_listed":1,"syntology":null},{"url":"/paper/reflectioncoder-learning-from-reflection","slug":"reflectioncoder-learning-from-reflection","title":"ReflectionCoder: Learning from Reflection Sequence for Enhanced One-off Code Generation","date":"2024-05-27","arxiv_id":"2405.17057","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reflectioncoder-learning-from-reflection#ran","syntology_url":"https://syntology.ai/paper/2405.17057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17057"}},"official":{"repos":["sensellm/reflectioncoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/rtl-repo-a-benchmark-for-evaluating-llms-on","slug":"rtl-repo-a-benchmark-for-evaluating-llms-on","title":"RTL-Repo: A Benchmark for Evaluating LLMs on Large-Scale RTL Design Projects","date":"2024-05-27","arxiv_id":"2405.17378","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-via-program-generation","slug":"learning-to-reason-via-program-generation","title":"Learning to Reason via Program Generation, Emulation, and Search","date":"2024-05-25","arxiv_id":"2405.16337","repositories_listed":1,"syntology":null},{"url":"/paper/chatgpt-code-detection-techniques-for","slug":"chatgpt-code-detection-techniques-for","title":"ChatGPT Code Detection: Techniques for Uncovering the Source of Code","date":"2024-05-24","arxiv_id":"2405.15512","repositories_listed":1,"syntology":null},{"url":"/paper/generating-code-world-models-with-large","slug":"generating-code-world-models-with-large","title":"Generating Code World Models with Large Language Models Guided by Monte Carlo Tree Search","date":"2024-05-24","arxiv_id":"2405.15383","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-code-world-models-with-large#ran","syntology_url":"https://syntology.ai/paper/2405.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15383"}},"official":{"repos":["nicoladainese96/code-world-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-tuning-strikes-back-customizing","slug":"prompt-tuning-strikes-back-customizing","title":"Prompt Tuning Strikes Back: Customizing Foundation Models with Low-Rank Prompt Adaptation","date":"2024-05-24","arxiv_id":"2405.15282","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompt-tuning-strikes-back-customizing#ran","syntology_url":"https://syntology.ai/paper/2405.15282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15282"}},"official":{"repos":["jabhinav/prompt-tuning-strikes-back-with-lopa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/analogcoder-analog-circuit-design-via","slug":"analogcoder-analog-circuit-design-via","title":"AnalogCoder: Analog Circuit Design via Training-Free Code Generation","date":"2024-05-23","arxiv_id":"2405.14918","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analogcoder-analog-circuit-design-via#ran","syntology_url":"https://syntology.ai/paper/2405.14918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14918"}},"official":{"repos":["laiyao1/AnalogCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autocoder-enhancing-code-large-language-model","slug":"autocoder-enhancing-code-large-language-model","title":"AutoCoder: Enhancing Code Large Language Model with \\textsc{AIEV-Instruct}","date":"2024-05-23","arxiv_id":"2405.14906","repositories_listed":1,"syntology":null},{"url":"/paper/comera-computing-and-memory-efficient","slug":"comera-computing-and-memory-efficient","title":"CoMERA: Computing- and Memory-Efficient Training via Rank-Adaptive Tensor Optimization","date":"2024-05-23","arxiv_id":"2405.14377","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/comera-computing-and-memory-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.14377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14377"}},"official":{"repos":["ziyangjoy/comera"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-github-issues-be-solved-with-tree-of","slug":"can-github-issues-be-solved-with-tree-of","title":"Can Github issues be solved with Tree Of Thoughts?","date":"2024-05-20","arxiv_id":"2405.13057","repositories_listed":1,"syntology":null},{"url":"/paper/mhpp-exploring-the-capabilities-and","slug":"mhpp-exploring-the-capabilities-and","title":"MHPP: Exploring the Capabilities and Limitations of Language Models Beyond Basic Code Generation","date":"2024-05-19","arxiv_id":"2405.11430","repositories_listed":1,"syntology":null},{"url":"/paper/documint-docstring-generation-for-python","slug":"documint-docstring-generation-for-python","title":"DocuMint: Docstring Generation for Python using Small Language Models","date":"2024-05-16","arxiv_id":"2405.10243","repositories_listed":1,"syntology":null},{"url":"/paper/llms-and-the-future-of-chip-design-unveiling","slug":"llms-and-the-future-of-chip-design-unveiling","title":"LLMs and the Future of Chip Design: Unveiling Security Risks and Building Trust","date":"2024-05-11","arxiv_id":"2405.07061","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-execution-based-evaluation-for","slug":"tackling-execution-based-evaluation-for","title":"Execution-Based Evaluation of Natural Language to Bash and PowerShell for Incident Remediation","date":"2024-05-10","arxiv_id":"2405.06807","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-synergize-with","slug":"large-language-models-synergize-with","title":"Large Language Models Synergize with Automated Machine Learning","date":"2024-05-06","arxiv_id":"2405.03727","repositories_listed":1,"syntology":null},{"url":"/paper/codehalu-code-hallucinations-in-llms-driven","slug":"codehalu-code-hallucinations-in-llms-driven","title":"CodeHalu: Investigating Code Hallucinations in LLMs via Execution-based Verification","date":"2024-04-30","arxiv_id":"2405.00253","repositories_listed":1,"syntology":null},{"url":"/paper/do-neutral-prompts-produce-insecure-code","slug":"do-neutral-prompts-produce-insecure-code","title":"How secure is AI-generated Code: A Large-Scale Comparison of Large Language Models","date":"2024-04-29","arxiv_id":"2404.18353","repositories_listed":1,"syntology":null},{"url":"/paper/pecc-problem-extraction-and-coding-challenges","slug":"pecc-problem-extraction-and-coding-challenges","title":"PECC: Problem Extraction and Coding Challenges","date":"2024-04-29","arxiv_id":"2404.18766","repositories_listed":1,"syntology":null},{"url":"/paper/ai-coders-are-among-us-rethinking-programming","slug":"ai-coders-are-among-us-rethinking-programming","title":"AI Coders Are Among Us: Rethinking Programming Language Grammar Towards Efficient Code Generation","date":"2024-04-25","arxiv_id":"2404.16333","repositories_listed":1,"syntology":null},{"url":"/paper/codeip-a-grammar-guided-multi-bit-watermark","slug":"codeip-a-grammar-guided-multi-bit-watermark","title":"CodeIP: A Grammar-Guided Multi-Bit Watermark for Large Language Models of Code","date":"2024-04-24","arxiv_id":"2404.15639","repositories_listed":1,"syntology":null},{"url":"/paper/class-level-code-generation-from-natural","slug":"class-level-code-generation-from-natural","title":"Class-Level Code Generation from Natural Language Using Iterative, Tool-Enhanced Reasoning over Repository","date":"2024-04-22","arxiv_id":"2405.01573","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/class-level-code-generation-from-natural#ran","syntology_url":"https://syntology.ai/paper/2405.01573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01573"}},"official":null}},{"url":"/paper/towards-smallers-faster-decoder-only","slug":"towards-smallers-faster-decoder-only","title":"Towards smaller, faster decoder-only transformers: Architectural variants and their implications","date":"2024-04-22","arxiv_id":"2404.14462","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-as-test-case-generators","slug":"large-language-models-as-test-case-generators","title":"Large Language Models as Test Case Generators: Performance Evaluation and Enhancement","date":"2024-04-20","arxiv_id":"2404.13340","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-as-test-case-generators#ran","syntology_url":"https://syntology.ai/paper/2404.13340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13340"}},"official":null}},{"url":"/paper/is-dpo-superior-to-ppo-for-llm-alignment-a","slug":"is-dpo-superior-to-ppo-for-llm-alignment-a","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","date":"2024-04-16","arxiv_id":"2404.10719","repositories_listed":1,"syntology":null},{"url":"/paper/comments-as-natural-logic-pivots-improve-code","slug":"comments-as-natural-logic-pivots-improve-code","title":"Comments as Natural Logic Pivots: Improve Code Generation via Comment Perspective","date":"2024-04-11","arxiv_id":"2404.07549","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-the-performance-of-large-language","slug":"analyzing-the-performance-of-large-language","title":"Analyzing the Performance of Large Language Models on Code Summarization","date":"2024-04-10","arxiv_id":"2404.08018","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-human-evaluation-of-large","slug":"sample-efficient-human-evaluation-of-large","title":"Sample-Efficient Human Evaluation of Large Language Models via Maximum Discrepancy Competition","date":"2024-04-10","arxiv_id":"2404.08008","repositories_listed":1,"syntology":null},{"url":"/paper/xiwu-a-basis-flexible-and-learnable-llm-for","slug":"xiwu-a-basis-flexible-and-learnable-llm-for","title":"Xiwu: A Basis Flexible and Learnable LLM for High Energy Physics","date":"2024-04-08","arxiv_id":"2404.08001","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-into-misuse-of-java-security","slug":"an-investigation-into-misuse-of-java-security","title":"An Investigation into Misuse of Java Security APIs by Large Language Models","date":"2024-04-04","arxiv_id":"2404.03823","repositories_listed":1,"syntology":null},{"url":"/paper/codeeditorbench-evaluating-code-editing","slug":"codeeditorbench-evaluating-code-editing","title":"CodeEditorBench: Evaluating Code Editing Capability of Large Language Models","date":"2024-04-04","arxiv_id":"2404.03543","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codeeditorbench-evaluating-code-editing#ran","syntology_url":"https://syntology.ai/paper/2404.03543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03543"}},"official":null}},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-organized-agents-a-llm-multi-agent","slug":"self-organized-agents-a-llm-multi-agent","title":"Self-Organized Agents: A LLM Multi-Agent Framework toward Ultra Large-Scale Code Generation and Optimization","date":"2024-04-02","arxiv_id":"2404.02183","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-organized-agents-a-llm-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2404.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02183"}},"official":{"repos":["tsukushiai/self-organized-agent"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evocodebench-an-evolving-code-generation","slug":"evocodebench-an-evolving-code-generation","title":"EvoCodeBench: An Evolving Code Generation Benchmark Aligned with Real-World Code Repositories","date":"2024-03-31","arxiv_id":"2404.00599","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 3 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evocodebench-an-evolving-code-generation#ran","syntology_url":"https://syntology.ai/paper/2404.00599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00599"}},"official":{"repos":["seketeam/evocodebench"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-larger-the-better-improved-llm-code","slug":"the-larger-the-better-improved-llm-code","title":"The Larger the Better? Improved LLM Code-Generation via Budget Reallocation","date":"2024-03-31","arxiv_id":"2404.00725","repositories_listed":1,"syntology":null},{"url":"/paper/top-leaderboard-ranking-top-coding","slug":"top-leaderboard-ranking-top-coding","title":"Top Leaderboard Ranking = Top Coding Proficiency, Always? EvoEval: Evolving Coding Benchmarks via LLM","date":"2024-03-28","arxiv_id":"2403.19114","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/top-leaderboard-ranking-top-coding#ran","syntology_url":"https://syntology.ai/paper/2403.19114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19114"}},"official":{"repos":["evo-eval/evoeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cycle-learning-to-self-refine-the-code","slug":"cycle-learning-to-self-refine-the-code","title":"CYCLE: Learning to Self-Refine the Code Generation","date":"2024-03-27","arxiv_id":"2403.18746","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-refinement-of-project-level-code","slug":"iterative-refinement-of-project-level-code","title":"Iterative Refinement of Project-Level Code Context for Precise Code Generation with Compiler Feedback","date":"2024-03-25","arxiv_id":"2403.16792","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-based-aesthetic-qr-code-generation","slug":"diffusion-based-aesthetic-qr-code-generation","title":"Diffusion-based Aesthetic QR Code Generation via Scanning-Robust Perceptual Guidance","date":"2024-03-23","arxiv_id":"2403.15878","repositories_listed":1,"syntology":null},{"url":"/paper/from-words-to-routes-applying-large-language","slug":"from-words-to-routes-applying-large-language","title":"Can Large Language Models Solve Robot Routing?","date":"2024-03-16","arxiv_id":"2403.10795","repositories_listed":1,"syntology":null},{"url":"/paper/autodev-automated-ai-driven-development","slug":"autodev-automated-ai-driven-development","title":"AutoDev: Automated AI-Driven Development","date":"2024-03-13","arxiv_id":"2403.08299","repositories_listed":1,"syntology":null},{"url":"/paper/bugs-in-large-language-models-generated-code","slug":"bugs-in-large-language-models-generated-code","title":"Bugs in Large Language Models Generated Code: An Empirical Study","date":"2024-03-13","arxiv_id":"2403.08937","repositories_listed":1,"syntology":null},{"url":"/paper/cleanagent-automating-data-standardization","slug":"cleanagent-automating-data-standardization","title":"CleanAgent: Automating Data Standardization with LLM-based Agents","date":"2024-03-13","arxiv_id":"2403.08291","repositories_listed":1,"syntology":null},{"url":"/paper/branch-train-mix-mixing-expert-llms-into-a","slug":"branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","arxiv_id":"2403.07816","repositories_listed":1,"syntology":null},{"url":"/paper/knowcoder-coding-structured-knowledge-into","slug":"knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","arxiv_id":"2403.07969","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-generation-of-python-programs-using","slug":"automatic-generation-of-python-programs-using","title":"Automatic Generation of Python Programs Using Context-Free Grammars","date":"2024-03-11","arxiv_id":"2403.06503","repositories_listed":1,"syntology":null},{"url":"/paper/text2qr-harmonizing-aesthetic-customization","slug":"text2qr-harmonizing-aesthetic-customization","title":"Text2QR: Harmonizing Aesthetic Customization and Scanning Robustness for Text-Guided QR Code Generation","date":"2024-03-11","arxiv_id":"2403.06452","repositories_listed":1,"syntology":null},{"url":"/paper/unisparse-an-intermediate-language-for","slug":"unisparse-an-intermediate-language-for","title":"UniSparse: An Intermediate Language for General Sparse Format Customization","date":"2024-03-09","arxiv_id":"2403.05802","repositories_listed":1,"syntology":null},{"url":"/paper/gemini-1-5-unlocking-multimodal-understanding","slug":"gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","arxiv_id":"2403.05530","repositories_listed":1,"syntology":null},{"url":"/paper/rat-retrieval-augmented-thoughts-elicit","slug":"rat-retrieval-augmented-thoughts-elicit","title":"RAT: Retrieval Augmented Thoughts Elicit Context-Aware Reasoning in Long-Horizon Generation","date":"2024-03-08","arxiv_id":"2403.05313","repositories_listed":1,"syntology":null},{"url":"/paper/ircoder-intermediate-representations-make","slug":"ircoder-intermediate-representations-make","title":"IRCoder: Intermediate Representations Make Language Models Robust Multilingual Code Generators","date":"2024-03-06","arxiv_id":"2403.03894","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-contamination-in-evaluating-code","slug":"quantifying-contamination-in-evaluating-code","title":"Quantifying Contamination in Evaluating Code Generation Capabilities of Language Models","date":"2024-03-06","arxiv_id":"2403.04811","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-contamination-in-evaluating-code#ran","syntology_url":"https://syntology.ai/paper/2403.04811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04811"}},"official":{"repos":["yale-nlp/code-llm-contamination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/daco-towards-application-driven-and","slug":"daco-towards-application-driven-and","title":"DACO: Towards Application-Driven and Comprehensive Data Analysis via Code Generation","date":"2024-03-04","arxiv_id":"2403.02528","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/daco-towards-application-driven-and#ran","syntology_url":"https://syntology.ai/paper/2403.02528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02528"}},"official":{"repos":["shirley-wu/daco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-llm-code-generation-with-grammar","slug":"improving-llm-code-generation-with-grammar","title":"SynCode: LLM Generation with Grammar Augmentation","date":"2024-03-03","arxiv_id":"2403.01632","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/improving-llm-code-generation-with-grammar#ran","syntology_url":"https://syntology.ai/paper/2403.01632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01632"}},"official":{"repos":["uiuc-focal-lab/syncode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/open-assistant-toolkit-version-2","slug":"open-assistant-toolkit-version-2","title":"Open Assistant Toolkit -- version 2","date":"2024-03-01","arxiv_id":"2403.00586","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-assistant-toolkit-version-2#ran","syntology_url":"https://syntology.ai/paper/2403.00586","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00586"}},"official":{"repos":["grill-lab/oat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-pre-trained-language-models-for-3","slug":"leveraging-pre-trained-language-models-for-3","title":"Leveraging pre-trained language models for code generation","date":"2024-02-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/data-interpreter-an-llm-agent-for-data","slug":"data-interpreter-an-llm-agent-for-data","title":"Data Interpreter: An LLM Agent For Data Science","date":"2024-02-28","arxiv_id":"2402.18679","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-information-refinement-training","slug":"unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","arxiv_id":"2402.18150","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-information-refinement-training#ran","syntology_url":"https://syntology.ai/paper/2402.18150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18150"}},"official":{"repos":["xsc1234/info-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-data-science-agents","slug":"benchmarking-data-science-agents","title":"Benchmarking Data Science Agents","date":"2024-02-27","arxiv_id":"2402.17168","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-science-agents#ran","syntology_url":"https://syntology.ai/paper/2402.17168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17168"}},"official":{"repos":["metacopilot/dseval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ds-agent-automated-data-science-by-empowering","slug":"ds-agent-automated-data-science-by-empowering","title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","date":"2024-02-27","arxiv_id":"2402.17453","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ds-agent-automated-data-science-by-empowering#ran","syntology_url":"https://syntology.ai/paper/2402.17453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17453"}},"official":{"repos":["guosyjlu/ds-agent"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"7c9f2a6d28511d567a090f386a770f9e4a8449e0af5994243d9287117682dc0d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}