{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/humaneval/papers/3","list_of":"/task/humaneval","task":"HumanEval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,264],"of":264,"counts":{"archive_papers_tagged":264,"with_a_code_link":135,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":264,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":65,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":65,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/humaneval","prev":"/task/humaneval/papers/2","next":null,"papers":[{"url":null,"slug":"context-augmented-code-generation-using","title":"Context-Augmented Code Generation Using Programming Knowledge Graphs","date":"2024-10-09","arxiv_id":"2410.18251","repositories_listed":0,"syntology":null},{"url":null,"slug":"aime-ai-system-optimization-via-multiple-llm","title":"AIME: AI System Optimization via Multiple LLM Evaluators","date":"2024-10-04","arxiv_id":"2410.03131","repositories_listed":0,"syntology":null},{"url":null,"slug":"selection-of-prompt-engineering-techniques","title":"Selection of Prompt Engineering Techniques for Code Generation through Predicting Code Complexity","date":"2024-09-24","arxiv_id":"2409.16416","repositories_listed":0,"syntology":null},{"url":null,"slug":"grin-gradient-informed-moe","title":"GRIN: GRadient-INformed MoE","date":"2024-09-18","arxiv_id":"2409.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinkmcts-refining-erroneous-thoughts-in","title":"RethinkMCTS: Refining Erroneous Thoughts in Monte Carlo Tree Search for Code Generation","date":"2024-09-15","arxiv_id":"2409.09584","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-critical-planning-step-learning-boosts","title":"CPL: Critical Plan Step Learning Boosts LLM Generalization in Reasoning Tasks","date":"2024-09-13","arxiv_id":"2409.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathbb-uscd-improving-code-generation-of-llms","title":"$\\mathbb{USCD}$: Improving Code Generation of LLMs by Uncertainty-Aware Selective Contrastive Decoding","date":"2024-09-09","arxiv_id":"2409.05923","repositories_listed":0,"syntology":null},{"url":null,"slug":"arctic-snowcoder-demystifying-high-quality","title":"Arctic-SnowCoder: Demystifying High-Quality Data in Code Pretraining","date":"2024-09-03","arxiv_id":"2409.02326","repositories_listed":0,"syntology":null},{"url":null,"slug":"cruxeval-x-a-benchmark-for-multilingual-code","title":"CRUXEval-X: A Benchmark for Multilingual Code Reasoning, Understanding and Execution","date":"2024-08-23","arxiv_id":"2408.13001","repositories_listed":0,"syntology":null},{"url":null,"slug":"domaineval-an-auto-constructed-benchmark-for","title":"DOMAINEVAL: An Auto-Constructed Benchmark for Multi-Domain Code Generation","date":"2024-08-23","arxiv_id":"2408.13204","repositories_listed":0,"syntology":null},{"url":null,"slug":"autotest-evolutionary-code-solution-selection","title":"AutoTest: Evolutionary Code Solution Selection with Test Cases","date":"2024-08-22","arxiv_id":"2408.12125","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-distillation-from-strong-to-weak","title":"Concept Distillation from Strong to Weak Models via Hypotheses-to-Theories Prompting","date":"2024-08-18","arxiv_id":"2408.09365","repositories_listed":0,"syntology":null},{"url":null,"slug":"threshold-filtering-packing-for-supervised","title":"Threshold Filtering Packing for Supervised Fine-Tuning: Training Related Samples within Packs","date":"2024-08-18","arxiv_id":"2408.09327","repositories_listed":0,"syntology":null},{"url":null,"slug":"codemirage-hallucinations-in-code-generated","title":"CodeMirage: Hallucinations in Code Generated by Large Language Models","date":"2024-08-14","arxiv_id":"2408.08333","repositories_listed":0,"syntology":null},{"url":null,"slug":"crest-effectively-compacting-a-datastore-for","title":"CREST: Effectively Compacting a Datastore For Retrieval-Based Speculative Decoding","date":"2024-08-08","arxiv_id":"2408.04678","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21227","title":"TaskEval: Assessing Difficulty of Code Generation Tasks for Large Language Models","date":"2024-07-30","arxiv_id":"2407.21227","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-flow-matching","title":"Discrete Flow Matching","date":"2024-07-22","arxiv_id":"2407.15595","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-your-model-assessing-the-impact-of","title":"MaPPing Your Model: Assessing the Impact of Adversarial Attacks on LLM-based Programming Assistants","date":"2024-07-12","arxiv_id":"2407.11072","repositories_listed":0,"syntology":null},{"url":null,"slug":"brevity-is-the-soul-of-wit-pruning-long-files","title":"Brevity is the soul of wit: Pruning long files for code generation","date":"2024-06-29","arxiv_id":"2407.00434","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-large-language-model-aided-program","title":"Towards Large Language Model Aided Program Refinement","date":"2024-06-26","arxiv_id":"2406.18616","repositories_listed":0,"syntology":null},{"url":null,"slug":"qiskit-humaneval-an-evaluation-benchmark-for","title":"Qiskit HumanEval: An Evaluation Benchmark For Quantum Code Generative Models","date":"2024-06-20","arxiv_id":"2406.14712","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-optimise-self-generated-preference-data","title":"Code-Optimise: Self-Generated Preference Data for Correctness and Efficiency","date":"2024-06-18","arxiv_id":"2406.12502","repositories_listed":0,"syntology":null},{"url":null,"slug":"reactor-mk-1-performances-mmlu-humaneval-and","title":"Reactor Mk.1 performances: MMLU, HumanEval and BBH test results","date":"2024-06-15","arxiv_id":"2406.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"plum-preference-learning-plus-test-cases","title":"$\\textbf{PLUM}$: Improving Code LMs with Execution-Guided On-Policy Preference Learning Driven By Synthetic Test Cases","date":"2024-06-11","arxiv_id":"2406.06887","repositories_listed":0,"syntology":null},{"url":null,"slug":"validating-llm-generated-programs-with","title":"Validating LLM-Generated Programs with Metamorphic Prompt Testing","date":"2024-06-11","arxiv_id":"2406.06864","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-your-data-spark-joy-performance-gains","title":"Does your data spark joy? Performance gains from domain upsampling at the end of training","date":"2024-06-05","arxiv_id":"2406.03476","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-meets-consensus-unleashing","title":"Divide-and-Conquer Meets Consensus: Unleashing the Power of Functions in Code Generation","date":"2024-05-30","arxiv_id":"2405.20092","repositories_listed":0,"syntology":null},{"url":null,"slug":"specdec-boosting-speculative-decoding-via","title":"SpecDec++: Boosting Speculative Decoding via Adaptive Candidate Lengths","date":"2024-05-30","arxiv_id":"2405.19715","repositories_listed":0,"syntology":null},{"url":null,"slug":"kotlin-ml-pack-technical-report","title":"Kotlin ML Pack: Technical Report","date":"2024-05-29","arxiv_id":"2405.19250","repositories_listed":0,"syntology":null},{"url":null,"slug":"qiskit-code-assistant-training-llms-for","title":"Qiskit Code Assistant: Training LLMs for generating Quantum Computing Code","date":"2024-05-29","arxiv_id":"2405.19495","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-limitations-of-embedding-based-methods","title":"On the Limitations of Embedding Based Methods for Measuring Functional Correctness for Code Generation","date":"2024-04-26","arxiv_id":"2405.01580","repositories_listed":0,"syntology":null},{"url":null,"slug":"bass-batched-attention-optimized-speculative","title":"BASS: Batched Attention-optimized Speculative Sampling","date":"2024-04-24","arxiv_id":"2404.15778","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-teaching-large-language-models-to-reason","title":"NExT: Teaching Large Language Models to Reason about Code Execution","date":"2024-04-23","arxiv_id":"2404.14662","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-cost-language-models-survey-and","title":"Low-Cost Language Models: Survey and Performance Evaluation on Python Code Generation","date":"2024-04-17","arxiv_id":"2404.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-and-evaluating-hallucinations-in","title":"Exploring and Evaluating Hallucinations in LLM-Powered Code Generation","date":"2024-04-01","arxiv_id":"2404.00971","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-large-language-models-with-runtime","title":"Reasoning Runtime Behavior of a Program with LLM: How Far Are We?","date":"2024-03-25","arxiv_id":"2403.16437","repositories_listed":0,"syntology":null},{"url":null,"slug":"codeshell-technical-report","title":"CodeShell Technical Report","date":"2024-03-23","arxiv_id":"2403.15747","repositories_listed":0,"syntology":null},{"url":"/paper/when-llm-based-code-generation-meets-the","slug":"when-llm-based-code-generation-meets-the","title":"SOEN-101: Code Generation by Emulating Software Process Models Using Large Language Model Agents","date":"2024-03-23","arxiv_id":"2403.15852","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-three-phases-sft-hybrid-model-integrated","title":"CodingTeachLLM: Empowering LLM's Coding Ability via AST Prior Knowledge","date":"2024-03-13","arxiv_id":"2403.15426","repositories_listed":0,"syntology":null},{"url":null,"slug":"software-vulnerability-and-functionality","title":"Software Vulnerability and Functionality Assessment using LLMs","date":"2024-03-13","arxiv_id":"2403.08429","repositories_listed":0,"syntology":null},{"url":null,"slug":"livecodebench-holistic-and-contamination-free","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","date":"2024-03-12","arxiv_id":"2403.07974","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-driven-development-for-code-generation","title":"Test-Driven Development for Code Generation","date":"2024-02-21","arxiv_id":"2402.13521","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-how-to-ask-cycle-consistency-refines","title":"Learning How To Ask: Cycle-Consistency Refines Prompts in Multimodal Foundation Models","date":"2024-02-13","arxiv_id":"2402.08756","repositories_listed":0,"syntology":null},{"url":null,"slug":"nofuneval-funny-how-code-lms-falter-on","title":"NoFunEval: Funny How Code LMs Falter on Requirements Beyond Functional Correctness","date":"2024-01-29","arxiv_id":"2401.15963","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutation-based-consistency-testing-for","title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","date":"2024-01-11","arxiv_id":"2401.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"boldly-going-where-no-benchmark-has-gone","title":"PythonSaga: Redefining the Benchmark to Evaluate Code Generating LLMs","date":"2024-01-08","arxiv_id":"2401.03855","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-repository-level-prompting-for","title":"A Review of Repository Level Prompting for LLMs","date":"2023-12-15","arxiv_id":"2312.10101","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-data-quality-via-synthetic","title":"Decoding Data Quality via Synthetic Corruptions: Embedding-guided Pruning of Code Data","date":"2023-12-05","arxiv_id":"2312.02418","repositories_listed":0,"syntology":null},{"url":null,"slug":"past-as-a-guide-leveraging-retrospective","title":"Past as a Guide: Leveraging Retrospective Learning for Python Code Completion","date":"2023-11-13","arxiv_id":"2311.07635","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-code-semantic-and-llms-semantic","title":"Bridging Code Semantic and LLMs: Semantic Chain-of-Thought Prompting for Code Generation","date":"2023-10-16","arxiv_id":"2310.10698","repositories_listed":0,"syntology":null},{"url":null,"slug":"codefuse-13b-a-pretrained-multi-lingual-code","title":"CodeFuse-13B: A Pretrained Multi-lingual Code Large Language Model","date":"2023-10-10","arxiv_id":"2310.06266","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-program-testing-ability-of-large-language","title":"The Program Testing Ability of Large Language Models for Code","date":"2023-10-09","arxiv_id":"2310.05727","repositories_listed":0,"syntology":null},{"url":null,"slug":"lord-low-rank-decomposition-of-monolingual","title":"LORD: Low Rank Decomposition Of Monolingual Code LLMs For One-Shot Compression","date":"2023-09-25","arxiv_id":"2309.14021","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-programming-languages-boost-each-other","title":"Can Programming Languages Boost Each Other via Instruction Tuning?","date":"2023-08-31","arxiv_id":"2308.16824","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecot-and-beyond-learning-to-program-and","title":"CodeCoT: Tackling Code Syntax Errors in CoT Reasoning for Code Generation","date":"2023-08-17","arxiv_id":"2308.08784","repositories_listed":0,"syntology":null},{"url":null,"slug":"pangu-coder2-boosting-large-language-models","title":"PanGu-Coder2: Boosting Large Language Models for Code with Ranking Feedback","date":"2023-07-27","arxiv_id":"2307.14936","repositories_listed":0,"syntology":null},{"url":null,"slug":"textbooks-are-all-you-need","title":"Textbooks Are All You Need","date":"2023-06-20","arxiv_id":"2306.11644","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfevolve-a-code-evolution-framework-via","title":"SelfEvolve: A Code Evolution Framework via Large Language Models","date":"2023-06-05","arxiv_id":"2306.02907","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-programming-thinking-in-large","title":"Structured Chain-of-Thought Prompting for Code Generation","date":"2023-05-11","arxiv_id":"2305.06599","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-code-generation","title":"Stochastic Code Generation","date":"2023-04-14","arxiv_id":"2304.08243","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-neural-model-meets-nl2code-a-survey","title":"Large Language Models Meet NL2Code: A Survey","date":"2022-12-19","arxiv_id":"2212.09420","repositories_listed":0,"syntology":null},{"url":"/paper/the-stack-3-tb-of-permissively-licensed","slug":"the-stack-3-tb-of-permissively-licensed","title":"The Stack: 3 TB of permissively licensed source code","date":"2022-11-20","arxiv_id":"2211.15533","repositories_listed":0,"syntology":null},{"url":null,"slug":"piloting-copilot-and-codex-hot-temperature","title":"Piloting Copilot, Codex, and StarCoder2: Hot Temperature, Cold Prompts, or Black Magic?","date":"2022-10-26","arxiv_id":"2210.14699","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-code-generation-via-test-driven","title":"Interactive Code Generation via Test-Driven User-Intent Formalization","date":"2022-08-11","arxiv_id":"2208.05950","repositories_listed":0,"syntology":null}],"record_sha256":"fcf6a204f45daffc06d5d032e8be27582f70fdd62e22e7c0008c976ca19c04b1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}