{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/humaneval/papers/2","list_of":"/task/humaneval","task":"HumanEval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":264,"counts":{"archive_papers_tagged":264,"with_a_code_link":135,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":264,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":65,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":65,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/humaneval","prev":"/task/humaneval","next":"/task/humaneval/papers/3","papers":[{"url":"/paper/humaneval-xl-a-multilingual-code-generation","slug":"humaneval-xl-a-multilingual-code-generation","title":"HumanEval-XL: A Multilingual Code Generation Benchmark for Cross-lingual Natural Language Generalization","date":"2024-02-26","arxiv_id":"2402.16694","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/humaneval-xl-a-multilingual-code-generation#ran","syntology_url":"https://syntology.ai/paper/2402.16694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16694"}},"official":{"repos":["FloatAI/HumanEval-XL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/ldb-a-large-language-model-debugger-via","slug":"ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","arxiv_id":"2402.16906","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ldb-a-large-language-model-debugger-via#ran","syntology_url":"https://syntology.ai/paper/2402.16906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16906"}},"official":{"repos":["floridsleeves/llmdebugger"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-or-memorization-data","slug":"generalization-or-memorization-data","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","date":"2024-02-24","arxiv_id":"2402.15938","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generalization-or-memorization-data#ran","syntology_url":"https://syntology.ai/paper/2402.15938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15938"}},"official":{"repos":["yihongdong/cdd-ted4llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opencodeinterpreter-integrating-code#ran","syntology_url":"https://syntology.ai/paper/2402.14658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14658"}},"official":null}},{"url":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/humaneval-on-latest-gpt-models-2024#ran","syntology_url":"https://syntology.ai/paper/2402.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14852"}},"official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dolphcoder-echo-locating-code-large-language","slug":"dolphcoder-echo-locating-code-large-language","title":"DolphCoder: Echo-Locating Code Large Language Models with Diverse and Multi-Objective Instruction Tuning","date":"2024-02-14","arxiv_id":"2402.09136","repositories_listed":1,"syntology":null},{"url":"/paper/getting-the-most-out-of-your-tokenizer-for","slug":"getting-the-most-out-of-your-tokenizer-for","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","date":"2024-02-01","arxiv_id":"2402.01035","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/getting-the-most-out-of-your-tokenizer-for#ran","syntology_url":"https://syntology.ai/paper/2402.01035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01035"}},"official":{"repos":["gautierdag/tokenizer-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-novel-approach-for-automatic-program-repair","slug":"a-novel-approach-for-automatic-program-repair","title":"A Novel Approach for Automatic Program Repair using Round-Trip Translation with Large Language Models","date":"2024-01-15","arxiv_id":"2401.07994","repositories_listed":1,"syntology":null},{"url":"/paper/oop-object-oriented-programming-evaluation","slug":"oop-object-oriented-programming-evaluation","title":"OOP: Object-Oriented Programming Evaluation Benchmark for Large Language Models","date":"2024-01-12","arxiv_id":"2401.06628","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/oop-object-oriented-programming-evaluation#ran","syntology_url":"https://syntology.ai/paper/2401.06628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06628"}},"official":{"repos":["alphadl/oop-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cruxeval-a-benchmark-for-code-reasoning","slug":"cruxeval-a-benchmark-for-code-reasoning","title":"CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution","date":"2024-01-05","arxiv_id":"2401.03065","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-fusion-advancing-prompt-evolution","slug":"instruction-fusion-advancing-prompt-evolution","title":"Instruction Fusion: Advancing Prompt Evolution through Hybridization","date":"2023-12-25","arxiv_id":"2312.15692","repositories_listed":1,"syntology":null},{"url":"/paper/repairllama-efficient-representations-and","slug":"repairllama-efficient-representations-and","title":"RepairLLaMA: Efficient Representations and Fine-Tuned Adapters for Program Repair","date":"2023-12-25","arxiv_id":"2312.15698","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/repairllama-efficient-representations-and#ran","syntology_url":"https://syntology.ai/paper/2312.15698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15698"}},"official":{"repos":["assert-kth/repairllama"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentcoder-multi-agent-based-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.13010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13010"}},"official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-benchmark-and-contamination-for#ran","syntology_url":"https://syntology.ai/paper/2311.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04850"}},"official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/personalised-distillation-empowering-open","slug":"personalised-distillation-empowering-open","title":"Personalised Distillation: Empowering Open-Sourced LLMs with Adaptive Learning for Code Generation","date":"2023-10-28","arxiv_id":"2310.18628","repositories_listed":1,"syntology":null},{"url":"/paper/crosscodeeval-a-diverse-and-multilingual","slug":"crosscodeeval-a-diverse-and-multilingual","title":"CrossCodeEval: A Diverse and Multilingual Benchmark for Cross-File Code Completion","date":"2023-10-17","arxiv_id":"2310.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crosscodeeval-a-diverse-and-multilingual#ran","syntology_url":"https://syntology.ai/paper/2310.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11248"}},"official":null}},{"url":"/paper/codechain-towards-modular-code-generation","slug":"codechain-towards-modular-code-generation","title":"CodeChain: Towards Modular Code Generation Through Chain of Self-revisions with Representative Sub-modules","date":"2023-10-13","arxiv_id":"2310.08992","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codechain-towards-modular-code-generation#ran","syntology_url":"https://syntology.ai/paper/2310.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08992"}},"official":{"repos":["SalesforceAIResearch/CodeChain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/enhancing-large-language-models-in-coding#ran","syntology_url":"https://syntology.ai/paper/2309.17272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17272"}},"official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-channel-dimensions-to-isolate","slug":"rethinking-channel-dimensions-to-isolate","title":"Rethinking Channel Dimensions to Isolate Outliers for Low-bit Weight Quantization of Large Language Models","date":"2023-09-27","arxiv_id":"2309.15531","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-channel-dimensions-to-isolate#ran","syntology_url":"https://syntology.ai/paper/2309.15531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15531"}},"official":{"repos":["johnheo/adadim-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-code-coverage-without-execution","slug":"predicting-code-coverage-without-execution","title":"Predicting Code Coverage without Execution","date":"2023-07-25","arxiv_id":"2307.13383","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-gpt-self-repair-for-code","slug":"demystifying-gpt-self-repair-for-code","title":"Is Self-Repair a Silver Bullet for Code Generation?","date":"2023-06-16","arxiv_id":"2306.09896","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-of-code-fail-at","slug":"large-language-models-of-code-fail-at","title":"Large Language Models of Code Fail at Completing Code with Potential Bugs","date":"2023-06-06","arxiv_id":"2306.03438","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-of-code-fail-at#ran","syntology_url":"https://syntology.ai/paper/2306.03438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03438"}},"official":{"repos":["amazon-science/buggy-code-completion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anpl-compiling-natural-programs-with","slug":"anpl-compiling-natural-programs-with","title":"ANPL: Towards Natural Programming with Interactive Decomposition","date":"2023-05-29","arxiv_id":"2305.18498","repositories_listed":1,"syntology":null},{"url":"/paper/leti-learning-to-generate-from-textual","slug":"leti-learning-to-generate-from-textual","title":"LeTI: Learning to Generate from Textual Interactions","date":"2023-05-17","arxiv_id":"2305.10314","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/leti-learning-to-generate-from-textual#ran","syntology_url":"https://syntology.ai/paper/2305.10314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10314"}},"official":{"repos":["xingyaoww/leti"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/self-edit-fault-aware-code-editor-for-code","slug":"self-edit-fault-aware-code-editor-for-code","title":"Self-Edit: Fault-Aware Code Editor for Code Generation","date":"2023-05-06","arxiv_id":"2305.04087","repositories_listed":1,"syntology":null},{"url":"/paper/is-your-code-generated-by-chatgpt-really-1","slug":"is-your-code-generated-by-chatgpt-really-1","title":"Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation","date":"2023-05-02","arxiv_id":"2305.01210","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-your-code-generated-by-chatgpt-really-1#ran","syntology_url":"https://syntology.ai/paper/2305.01210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01210"}},"official":{"repos":["evalplus/evalplus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-effectiveness-of-large-language","slug":"exploring-the-effectiveness-of-large-language","title":"Using Large Language Models to Generate JUnit Tests: An Empirical Study","date":"2023-04-30","arxiv_id":"2305.00418","repositories_listed":1,"syntology":null},{"url":"/paper/parsel-a-unified-natural-language-framework","slug":"parsel-a-unified-natural-language-framework","title":"Parsel: Algorithmic Reasoning with Language Models by Composing Decompositions","date":"2022-12-20","arxiv_id":"2212.10561","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parsel-a-unified-natural-language-framework#ran","syntology_url":"https://syntology.ai/paper/2212.10561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10561"}},"official":{"repos":["ezelikman/parsel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-how-fine-tuning-on-bimodal-data","slug":"evaluating-how-fine-tuning-on-bimodal-data","title":"Evaluating How Fine-tuning on Bimodal Data Effects Code Generation","date":"2022-11-15","arxiv_id":"2211.07842","repositories_listed":1,"syntology":null},{"url":"/paper/contragen-effective-contrastive-learning-for","slug":"contragen-effective-contrastive-learning-for","title":"ContraCLM: Contrastive Learning For Causal Language Model","date":"2022-10-03","arxiv_id":"2210.01185","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/contragen-effective-contrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2210.01185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01185"}},"official":null}},{"url":"/paper/a-scalable-and-extensible-approach-to","slug":"a-scalable-and-extensible-approach-to","title":"MultiPL-E: A Scalable and Extensible Approach to Benchmarking Neural Code Generation","date":"2022-08-17","arxiv_id":"2208.08227","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-scalable-and-extensible-approach-to#ran","syntology_url":"https://syntology.ai/paper/2208.08227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.08227"}},"official":{"repos":["nuprl/multipl-e"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codet-code-generation-with-generated-tests","slug":"codet-code-generation-with-generated-tests","title":"CodeT: Code Generation with Generated Tests","date":"2022-07-21","arxiv_id":"2207.10397","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codet-code-generation-with-generated-tests#ran","syntology_url":"https://syntology.ai/paper/2207.10397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10397"}},"official":{"repos":["microsoft/codet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fault-aware-neural-code-rankers","slug":"fault-aware-neural-code-rankers","title":"Fault-Aware Neural Code Rankers","date":"2022-06-04","arxiv_id":"2206.03865","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fault-aware-neural-code-rankers#ran","syntology_url":"https://syntology.ai/paper/2206.03865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03865"}},"official":{"repos":["microsoft/coderanker"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":null,"slug":"turning-the-tide-repository-based-code","title":"Turning the Tide: Repository-based Code Reflection","date":"2025-07-14","arxiv_id":"2507.09866","repositories_listed":0,"syntology":null},{"url":null,"slug":"sacl-understanding-and-combating-textual-bias","title":"SACL: Understanding and Combating Textual Bias in Code Retrieval with Semantic-Augmented Reranking and Localization","date":"2025-06-25","arxiv_id":"2506.20081","repositories_listed":0,"syntology":null},{"url":"/paper/plan-for-speed-dilated-scheduling-for-masked","slug":"plan-for-speed-dilated-scheduling-for-masked","title":"Plan for Speed -- Dilated Scheduling for Masked Diffusion Language Models","date":"2025-06-23","arxiv_id":"2506.19037","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plan-for-speed-dilated-scheduling-for-masked#ran","syntology_url":"https://syntology.ai/paper/2506.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19037"}},"official":null}},{"url":null,"slug":"guaranteed-guess-a-language-modeling-approach","title":"Guaranteed Guess: A Language Modeling Approach for CISC-to-RISC Transpilation with Testing Guarantees","date":"2025-06-17","arxiv_id":"2506.14606","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-mixer-coordinate-modular-lora-experts","title":"LoRA-Mixer: Coordinate Modular LoRA Experts Through Serial Attention Routing","date":"2025-06-17","arxiv_id":"2507.00029","repositories_listed":0,"syntology":null},{"url":null,"slug":"guideline-forest-experience-induced-multi","title":"Guideline Forest: Experience-Induced Multi-Guideline Reasoning with Stepwise Aggregation","date":"2025-06-09","arxiv_id":"2506.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"swifteval-developing-a-language-specific","title":"SwiftEval: Developing a Language-Specific Benchmark for LLM-generated Code Evaluation","date":"2025-05-30","arxiv_id":"2505.24324","repositories_listed":0,"syntology":null},{"url":"/paper/actor-critic-based-online-data-mixing-for","slug":"actor-critic-based-online-data-mixing-for","title":"Actor-Critic based Online Data Mixing For Language Model Pre-Training","date":"2025-05-29","arxiv_id":"2505.23878","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-critic-based-online-data-mixing-for#ran","syntology_url":"https://syntology.ai/paper/2505.23878","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23878"}},"official":null}},{"url":null,"slug":"enhancing-llm-based-code-generation-with","title":"Enhancing LLM-Based Code Generation with Complexity Metrics: A Feedback-Driven Approach","date":"2025-05-29","arxiv_id":"2505.23953","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-llm-as-judge-metric-for-bridging-the-gap","title":"An LLM-as-Judge Metric for Bridging the Gap with Human Evaluation in SE Tasks","date":"2025-05-27","arxiv_id":"2505.20854","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-large-language-models-for-code","title":"Evaluating Large Language Models for Code Review","date":"2025-05-26","arxiv_id":"2505.20206","repositories_listed":0,"syntology":null},{"url":null,"slug":"llada-1-5-variance-reduced-preference","title":"LLaDA 1.5: Variance-Reduced Preference Optimization for Large Language Diffusion Models","date":"2025-05-25","arxiv_id":"2505.19223","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-output-to-evaluation-does-raw","title":"From Output to Evaluation: Does Raw Instruction-Tuned Code LLMs Output Suffice for Fill-in-the-Middle Code Generation?","date":"2025-05-24","arxiv_id":"2505.18789","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-prompt-engineering-for-reinforcement","title":"Prior Prompt Engineering for Reinforcement Fine-Tuning","date":"2025-05-20","arxiv_id":"2505.14157","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-the-diffusion-chain-of-lateral","title":"Reinforcing the Diffusion Chain of Lateral Thought with Diffusion Language Models","date":"2025-05-15","arxiv_id":"2505.10446","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentioninfluence-adopting-attention-head","title":"AttentionInfluence: Adopting Attention Head Influence for Weak-to-Strong Pretraining Data Selection","date":"2025-05-12","arxiv_id":"2505.07293","repositories_listed":0,"syntology":null},{"url":null,"slug":"codemixbench-evaluating-large-language-models","title":"CodeMixBench: Evaluating Large Language Models on Code Generation with Code-Mixed Prompts","date":"2025-05-08","arxiv_id":"2505.05063","repositories_listed":0,"syntology":null},{"url":null,"slug":"memorization-or-interpolation-detecting-llm","title":"Memorization or Interpolation ? Detecting LLM Memorization through Input Perturbation Analysis","date":"2025-05-05","arxiv_id":"2505.03019","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-art-of-repair-optimizing-iterative","title":"The Art of Repair: Optimizing Iterative Program Repair with Instruction-Tuned Models","date":"2025-05-05","arxiv_id":"2505.02931","repositories_listed":0,"syntology":null},{"url":null,"slug":"arcs-agentic-retrieval-augmented-code","title":"ARCS: Agentic Retrieval-Augmented Code Synthesis with Iterative Refinement","date":"2025-04-29","arxiv_id":"2504.20434","repositories_listed":0,"syntology":null},{"url":null,"slug":"type-constrained-code-generation-with","title":"Type-Constrained Code Generation with Language Models","date":"2025-04-12","arxiv_id":"2504.09246","repositories_listed":0,"syntology":null},{"url":null,"slug":"opencodeinstruct-a-large-scale-instruction","title":"OpenCodeInstruct: A Large-scale Instruction Tuning Dataset for Code LLMs","date":"2025-04-05","arxiv_id":"2504.04030","repositories_listed":0,"syntology":null},{"url":null,"slug":"sustainable-llm-inference-for-edge-ai","title":"Sustainable LLM Inference for Edge AI: Evaluating Quantized LLMs for Energy Efficiency, Output Accuracy, and Inference Latency","date":"2025-04-04","arxiv_id":"2504.03360","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-enable-verification-in-mainstream","title":"Can LLMs Enable Verification in Mainstream Programming?","date":"2025-03-18","arxiv_id":"2503.14183","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-autonomous-programming-using-iterative","title":"Fully Autonomous Programming using Iterative Multi-Agent Debugging with Large Language Models","date":"2025-03-10","arxiv_id":"2503.07693","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-ai-models-in-software","title":"Benchmarking AI Models in Software Engineering: A Review, Search Tool, and Enhancement Protocol","date":"2025-03-07","arxiv_id":"2503.05860","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-based-code-representation-is-it-a","title":"Grammar-Based Code Representation: Is It a Worthy Pursuit for LLMs?","date":"2025-03-07","arxiv_id":"2503.05507","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-aware-task-arithmetic-disentangling","title":"Layer-Aware Task Arithmetic: Disentangling Task-Specific and Instruction-Following Knowledge","date":"2025-02-27","arxiv_id":"2502.20186","repositories_listed":0,"syntology":null},{"url":null,"slug":"isolating-language-coding-from-problem","title":"Isolating Language-Coding from Problem-Solving: Benchmarking LLMs with PseudoEval","date":"2025-02-26","arxiv_id":"2502.19149","repositories_listed":0,"syntology":null},{"url":null,"slug":"unitcoder-scalable-iterative-code-synthesis","title":"UnitCoder: Scalable Iterative Code Synthesis with Unit Test Guidance","date":"2025-02-17","arxiv_id":"2502.11460","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-guided-self-debugging","title":"Large Language Model Guided Self-Debugging Code Generation","date":"2025-02-05","arxiv_id":"2502.02928","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-as-logic-units-scaling-test-time","title":"Reasoning-as-Logic-Units: Scaling Test-Time Reasoning in Large Language Models Through Logic Unit Alignment","date":"2025-02-05","arxiv_id":"2502.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"acecoder-acing-coder-rl-via-automated-test","title":"ACECODER: Acing Coder RL via Automated Test-Case Synthesis","date":"2025-02-03","arxiv_id":"2502.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"importing-phantoms-measuring-llm-package","title":"Importing Phantoms: Measuring LLM Package Hallucination Vulnerabilities","date":"2025-01-31","arxiv_id":"2501.19012","repositories_listed":0,"syntology":null},{"url":"/paper/qualityflow-an-agentic-workflow-for-program","slug":"qualityflow-an-agentic-workflow-for-program","title":"QualityFlow: An Agentic Workflow for Program Synthesis Controlled by LLM Quality Checks","date":"2025-01-20","arxiv_id":"2501.17167","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-metamemory-mechanisms-for-enhanced","title":"Leveraging Metamemory Mechanisms for Enhanced Data-Free Code Generation in LLMs","date":"2025-01-14","arxiv_id":"2501.07892","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-code-generation-with-llms-a-multi","title":"Guided Code Generation with LLMs: A Multi-Agent Framework for Complex Code Tasks","date":"2025-01-11","arxiv_id":"2501.06625","repositories_listed":0,"syntology":null},{"url":null,"slug":"dafny-as-verification-aware-intermediate","title":"Dafny as Verification-Aware Intermediate Language for Code Generation","date":"2025-01-10","arxiv_id":"2501.06283","repositories_listed":0,"syntology":null},{"url":null,"slug":"infifusion-a-unified-framework-for-enhanced","title":"InfiFusion: A Unified Framework for Enhanced Cross-Model Reasoning via LLM Fusion","date":"2025-01-06","arxiv_id":"2501.02795","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-scaling-of-unit-tests-for-code-reward","title":"Dynamic Scaling of Unit Tests for Code Reward Modeling","date":"2025-01-02","arxiv_id":"2501.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"secbench-a-comprehensive-multi-dimensional","title":"SecBench: A Comprehensive Multi-Dimensional Benchmarking Dataset for LLMs in Cybersecurity","date":"2024-12-30","arxiv_id":"2412.20787","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-before-running-efficient-code","title":"Thinking Before Running! Efficient Code Generation with Thorough Exploration and Optimal Refinement","date":"2024-12-30","arxiv_id":"2502.17442","repositories_listed":0,"syntology":null},{"url":null,"slug":"dovetail-a-cpu-gpu-heterogeneous-speculative","title":"Dovetail: A CPU/GPU Heterogeneous Speculative Decoding for LLM inference","date":"2024-12-25","arxiv_id":"2412.18934","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-aware-fine-tuning-for-best-of-n","title":"Inference-Aware Fine-Tuning for Best-of-N Sampling in Large Language Models","date":"2024-12-18","arxiv_id":"2412.15287","repositories_listed":0,"syntology":null},{"url":null,"slug":"falcon-faster-and-parallel-inference-of-large","title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","date":"2024-12-17","arxiv_id":"2412.12639","repositories_listed":0,"syntology":null},{"url":null,"slug":"perc-plan-as-query-example-retrieval-for","title":"PERC: Plan-As-Query Example Retrieval for Underrepresented Code Generation","date":"2024-12-17","arxiv_id":"2412.12447","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-via-self-iterative-process","title":"Learning to Reason via Self-Iterative Process Feedback for Small Language Models","date":"2024-12-11","arxiv_id":"2412.08393","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphaverus-bootstrapping-formally-verified","title":"AlphaVerus: Bootstrapping Formally Verified Code Generation through Self-Improving Translation and Treefinement","date":"2024-12-09","arxiv_id":"2412.06176","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-few-shot-learning-help-llm-performance","title":"Does Few-Shot Learning Help LLM Performance in Code Synthesis?","date":"2024-12-03","arxiv_id":"2412.02906","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-data-leakage-in-humaneval-using","title":"Addressing Data Leakage in HumanEval Using Combinatorial Test Design","date":"2024-12-02","arxiv_id":"2412.01526","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-of-multilingual-code","title":"A Preliminary Study of Multilingual Code Language Models for Code Generation Task Using Translated Benchmarks","date":"2024-11-23","arxiv_id":"2411.15470","repositories_listed":0,"syntology":null},{"url":null,"slug":"dstc-direct-preference-learning-with-only","title":"DSTC: Direct Preference Learning with Only Self-Generated Tests and Code to Improve Code LMs","date":"2024-11-20","arxiv_id":"2411.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"valtest-automated-validation-of-language","title":"VALTEST: Automated Validation of Language Model Generated Test Cases","date":"2024-11-13","arxiv_id":"2411.08254","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesize-partition-then-adapt-eliciting","title":"Synthesize, Partition, then Adapt: Eliciting Diverse Samples from Foundation Models","date":"2024-11-11","arxiv_id":"2411.06722","repositories_listed":0,"syntology":null},{"url":null,"slug":"codetree-agent-guided-tree-search-for-code","title":"CodeTree: Agent-guided Tree Search for Code Generation with Large Language Models","date":"2024-11-07","arxiv_id":"2411.04329","repositories_listed":0,"syntology":null},{"url":null,"slug":"democraft-using-in-context-learning-to","title":"Demo-Craft: Using In-Context Learning to Improve Code Generation in Large Language Models","date":"2024-10-30","arxiv_id":"2411.00865","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-codellms-with-direct-preference","title":"Aligning CodeLLMs with Direct Preference Optimization","date":"2024-10-24","arxiv_id":"2410.18585","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dense-reward-understanding-the-gap","title":"Adaptive Dense Reward: Understanding the Gap Between Action and Reward Space in Alignment","date":"2024-10-23","arxiv_id":"2411.00809","repositories_listed":0,"syntology":null},{"url":null,"slug":"mojobench-language-modeling-and-benchmarks","title":"MojoBench: Language Modeling and Benchmarks for Mojo","date":"2024-10-23","arxiv_id":"2410.17736","repositories_listed":0,"syntology":null},{"url":null,"slug":"scattered-forest-search-smarter-code-space","title":"Scattered Forest Search: Smarter Code Space Exploration with LLMs","date":"2024-10-22","arxiv_id":"2411.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolving-multi-agent-collaboration","title":"Self-Evolving Multi-Agent Collaboration Networks for Software Development","date":"2024-10-22","arxiv_id":"2410.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-guided-search-for-efficient-program","title":"Semantic-guided Search for Efficient Program Repair with Large Language Models","date":"2024-10-22","arxiv_id":"2410.16655","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-explained-keywords-empower-large","title":"Self-Explained Keywords Empower Large Language Models for Code Generation","date":"2024-10-21","arxiv_id":"2410.15966","repositories_listed":0,"syntology":null},{"url":null,"slug":"celi-controller-embedded-language-model","title":"CELI: Controller-Embedded Language Model Interactions","date":"2024-10-18","arxiv_id":"2410.14627","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-designer-architecting-multi-agent","title":"G-Designer: Architecting Multi-agent Communication Topologies via Graph Neural Networks","date":"2024-10-15","arxiv_id":"2410.11782","repositories_listed":0,"syntology":null}],"record_sha256":"b5d4c0a9eafb0afa5bf4a056ba79f97110293937f408449053964fcfad02d152","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}