{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/12","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":17,"rows_per_page":100,"rows":[1101,1200],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/11","next":"/task/code-generation/papers/13","papers":[{"url":null,"slug":"tree-of-code-a-hybrid-approach-for-robust","title":"Tree-of-Code: A Hybrid Approach for Robust Complex Task Planning and Execution","date":"2024-12-18","arxiv_id":"2412.14212","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploratory-study-of-ml-sketches-and","title":"An Exploratory Study of ML Sketches and Visual Code Assistants","date":"2024-12-17","arxiv_id":"2412.13386","repositories_listed":0,"syntology":null},{"url":null,"slug":"analogxpert-automating-analog-topology","title":"AnalogXpert: Automating Analog Topology Synthesis by Incorporating Circuit Design Expertise into Large Language Models","date":"2024-12-17","arxiv_id":"2412.19824","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-programming-language-barrier","title":"Breaking the Programming Language Barrier: Multilingual Prompting to Empower Non-Native English Learners","date":"2024-12-17","arxiv_id":"2412.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"perc-plan-as-query-example-retrieval-for","title":"PERC: Plan-As-Query Example Retrieval for Underrepresented Code Generation","date":"2024-12-17","arxiv_id":"2412.12447","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed-cts-unleashing-the-power-of-tree-search","title":"Seed-CTS: Unleashing the Power of Tree Search for Superior Performance in Competitive Coding Tasks","date":"2024-12-17","arxiv_id":"2412.12544","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptv-leveraging-llm-powered-multi-agent","title":"CoopetitiveV: Leveraging LLM-powered Coopetitive Multi-Agent Prompting for High-quality Verilog Generation","date":"2024-12-15","arxiv_id":"2412.11014","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-ai-assisted-code-generation","title":"Optimizing AI-Assisted Code Generation","date":"2024-12-14","arxiv_id":"2412.10953","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogagent-an-auto-engagement-agent-for-code","title":"DialogAgent: An Auto-engagement Agent for Code Question Answering Data Production","date":"2024-12-11","arxiv_id":"2412.08069","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestromotif-skill-design-from-artificial","title":"MaestroMotif: Skill Design from Artificial Intelligence Feedback","date":"2024-12-11","arxiv_id":"2412.08542","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-llm-based-optimization-compilers-can","title":"Towards LLM-based optimization compilers. Can LLMs learn how to apply a single peephole optimization? Reasoning is all LLMs need!","date":"2024-12-11","arxiv_id":"2412.12163","repositories_listed":0,"syntology":null},{"url":null,"slug":"unseen-horizons-unveiling-the-real-capability","title":"Unseen Horizons: Unveiling the Real Capability of LLM Code Generation Beyond the Familiar","date":"2024-12-11","arxiv_id":"2412.08109","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-you-see-is-not-always-what-you-get-an","title":"What You See Is Not Always What You Get: An Empirical Study of Code Comprehension by Large Language Models","date":"2024-12-11","arxiv_id":"2412.08098","repositories_listed":0,"syntology":null},{"url":null,"slug":"arceak-an-automated-rule-checking-framework","title":"ARCEAK: An Automated Rule Checking Framework Enhanced with Architectural Knowledge","date":"2024-12-10","arxiv_id":"2501.14735","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoprep-natural-language-question-aware-data","title":"AutoPrep: Natural Language Question-Aware Data Preparation with a Multi-Agent Framework","date":"2024-12-10","arxiv_id":"2412.10422","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-coding-spot-understanding","title":"Exploring Coding Spot: Understanding Parametric Contributions to LLM Coding Performance","date":"2024-12-10","arxiv_id":"2412.07113","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphaverus-bootstrapping-formally-verified","title":"AlphaVerus: Bootstrapping Formally Verified Code Generation through Self-Improving Translation and Treefinement","date":"2024-12-09","arxiv_id":"2412.06176","repositories_listed":0,"syntology":null},{"url":null,"slug":"pyranet-a-multi-layered-hierarchical-dataset","title":"PyraNet: A Multi-Layered Hierarchical Dataset for Verilog","date":"2024-12-09","arxiv_id":"2412.06947","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-code-generation-with","title":"A Comparative Study on Code Generation with Transformers","date":"2024-12-07","arxiv_id":"2412.05749","repositories_listed":0,"syntology":null},{"url":null,"slug":"gee-ops-an-operator-knowledge-base-for","title":"GEE-OPs: An Operator Knowledge Base for Geospatial Code Generation on the Google Earth Engine Platform Powered by Large Language Models","date":"2024-12-07","arxiv_id":"2412.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-generation-and-runtime-techniques-for","title":"Code generation and runtime techniques for enabling data-efficient deep learning training on GPUs","date":"2024-12-06","arxiv_id":"2412.04747","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-cross-language-code-translation-via","title":"Enhancing Cross-Language Code Translation via Task-Specific Embedding Alignment in Retrieval-Augmented Generation","date":"2024-12-06","arxiv_id":"2412.05159","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-aligning-codellms-on-human","title":"Evaluating and Aligning CodeLLMs on Human Preference","date":"2024-12-06","arxiv_id":"2412.05210","repositories_listed":0,"syntology":null},{"url":null,"slug":"hivegen-hierarchical-llm-based-verilog","title":"HiVeGen -- Hierarchical LLM-based Verilog Generation for Scalable Chip Design","date":"2024-12-06","arxiv_id":"2412.05393","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypergraphos-a-meta-operating-system-for","title":"HyperGraphOS: A Meta Operating System for Science and Engineering","date":"2024-12-06","arxiv_id":"2412.04923","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-latex-code-generation-from","title":"Automated LaTeX Code Generation from Handwritten Math Expressions Using Vision Transformer","date":"2024-12-05","arxiv_id":"2412.03853","repositories_listed":0,"syntology":null},{"url":null,"slug":"bigdocs-an-open-and-permissively-licensed","title":"BigDocs: An Open and Permissively-Licensed Dataset for Training Multimodal Models on Document and Code Tasks","date":"2024-12-05","arxiv_id":"2412.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"if-you-can-t-use-them-recycle-them-optimizing","title":"If You Can't Use Them, Recycle Them: Optimizing Merging at Scale Mitigates Performance Tradeoffs","date":"2024-12-05","arxiv_id":"2412.04144","repositories_listed":0,"syntology":null},{"url":null,"slug":"potable-programming-standardly-on-table-based","title":"PoTable: Towards Systematic Thinking via Stage-oriented Plan-then-Execute Reasoning on Tables","date":"2024-12-05","arxiv_id":"2412.04272","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-different-large-language-model","title":"Survey of different Large Language Model Architectures: Trends, Benchmarks, and Challenges","date":"2024-12-04","arxiv_id":"2412.03220","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-framework-for-extensible","title":"A Multi-Agent Framework for Extensible Structured Text Generation in PLCs","date":"2024-12-03","arxiv_id":"2412.02410","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-few-shot-learning-help-llm-performance","title":"Does Few-Shot Learning Help LLM Performance in Code Synthesis?","date":"2024-12-03","arxiv_id":"2412.02906","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-agents-with-weakly-supervised","title":"Training Agents with Weakly Supervised Feedback from Large Language Models","date":"2024-11-29","arxiv_id":"2411.19547","repositories_listed":0,"syntology":null},{"url":null,"slug":"drc-coder-automated-drc-checker-code","title":"DRC-Coder: Automated DRC Checker Code Generation Using LLM Autonomous Agent","date":"2024-11-28","arxiv_id":"2412.05311","repositories_listed":0,"syntology":null},{"url":null,"slug":"face2qr-a-unified-framework-for-aesthetic","title":"Face2QR: A Unified Framework for Aesthetic, Face-Preserving, and Scannable QR Code Generation","date":"2024-11-28","arxiv_id":"2411.19246","repositories_listed":0,"syntology":null},{"url":null,"slug":"factchexcker-mitigating-measurement","title":"FactCheXcker: Mitigating Measurement Hallucinations in Chest X-ray Report Generation Models","date":"2024-11-27","arxiv_id":"2411.18672","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-deployment-of-untrusted-llms-reduces","title":"Adaptive Deployment of Untrusted LLMs Reduces Distributed Threats","date":"2024-11-26","arxiv_id":"2411.17693","repositories_listed":0,"syntology":null},{"url":null,"slug":"malmm-multi-agent-large-language-models-for","title":"MALMM: Multi-Agent Large Language Models for Zero-Shot Robotics Manipulation","date":"2024-11-26","arxiv_id":"2411.17636","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-of-multilingual-code","title":"A Preliminary Study of Multilingual Code Language Models for Code Generation Task Using Translated Benchmarks","date":"2024-11-23","arxiv_id":"2411.15470","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-mesh-with-me-generating-constructive","title":"Don't Mesh with Me: Generating Constructive Solid Geometry Instead of Meshes by Fine-Tuning a Code-Generation LLM","date":"2024-11-22","arxiv_id":"2411.15279","repositories_listed":0,"syntology":null},{"url":null,"slug":"eda-aware-rtl-generation-with-large-language","title":"EDA-Aware RTL Generation with Large Language Models","date":"2024-11-21","arxiv_id":"2412.04485","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-fuzzy-system-for-sequence","title":"Generative Fuzzy System for Sequence Generation","date":"2024-11-21","arxiv_id":"2411.13867","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-large-language-models-memorizing-bug","title":"Are Large Language Models Memorizing Bug Benchmarks?","date":"2024-11-20","arxiv_id":"2411.13323","repositories_listed":0,"syntology":null},{"url":null,"slug":"dstc-direct-preference-learning-with-only","title":"DSTC: Direct Preference Learning with Only Self-Generated Tests and Code to Improve Code LMs","date":"2024-11-20","arxiv_id":"2411.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-specification-driven-llm-based","title":"Towards Specification-Driven LLM-Based Generation of Embedded Automotive Software","date":"2024-11-20","arxiv_id":"2411.13269","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-historical-clinical-trial-data-with","title":"Unlocking Historical Clinical Trial Data with ALIGN: A Compositional Large Language Model System for Medical Coding","date":"2024-11-20","arxiv_id":"2411.13163","repositories_listed":0,"syntology":null},{"url":null,"slug":"libevolutioneval-a-benchmark-and-study-for","title":"LibEvolutionEval: A Benchmark and Study for Version-Specific Code Generation","date":"2024-11-19","arxiv_id":"2412.04478","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-programming-cop-empowering-large","title":"Chain-of-Programming (CoP) : Empowering Large Language Models for Geospatial Code Generation","date":"2024-11-16","arxiv_id":"2411.10753","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm4ds-evaluating-large-language-models-for","title":"LLM4DS: Evaluating Large Language Models for Data Science Code Generation","date":"2024-11-16","arxiv_id":"2411.11908","repositories_listed":0,"syntology":null},{"url":null,"slug":"see-saw-generative-mechanism-for-scalable","title":"See-Saw Generative Mechanism for Scalable Recursive Code Generation with Generative AI","date":"2024-11-16","arxiv_id":"2411.10861","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-prompt-formatting-have-any-impact-on-llm","title":"Does Prompt Formatting Have Any Impact on LLM Performance?","date":"2024-11-15","arxiv_id":"2411.10541","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-mmeval-a-parallel-multilingual-multitask","title":"P-MMEval: A Parallel Multilingual Multitask Benchmark for Consistent Evaluation of LLMs","date":"2024-11-14","arxiv_id":"2411.09116","repositories_listed":0,"syntology":null},{"url":null,"slug":"programming-with-ai-evaluating-chatgpt-gemini","title":"Programming with AI: Evaluating ChatGPT, Gemini, AlphaCode, and GitHub Copilot for Programmers","date":"2024-11-14","arxiv_id":"2411.09224","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-classification-of-open-source-ml","title":"Towards a Classification of Open-Source ML Models and Datasets for Software Engineering","date":"2024-11-14","arxiv_id":"2411.09683","repositories_listed":0,"syntology":null},{"url":null,"slug":"lynx-enabling-efficient-moe-inference-through","title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","date":"2024-11-13","arxiv_id":"2411.08982","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-of-ai-driven","title":"A Comprehensive Survey of AI-Driven Advancements and Techniques in Automated Program Repair and Code Generation","date":"2024-11-12","arxiv_id":"2411.07586","repositories_listed":0,"syntology":null},{"url":"/paper/spider-2-0-evaluating-language-models-on-real","slug":"spider-2-0-evaluating-language-models-on-real","title":"Spider 2.0: Evaluating Language Models on Real-World Enterprise Text-to-SQL Workflows","date":"2024-11-12","arxiv_id":"2411.07763","repositories_listed":0,"syntology":null},{"url":null,"slug":"pdc-dm-sft-a-road-for-llm-sql-bug-fix","title":"PDC & DM-SFT: A Road for LLM SQL Bug-Fix Enhancing","date":"2024-11-11","arxiv_id":"2411.06767","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesize-partition-then-adapt-eliciting","title":"Synthesize, Partition, then Adapt: Eliciting Diverse Samples from Foundation Models","date":"2024-11-11","arxiv_id":"2411.06722","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-answerability-of-queries-in","title":"Assessing the Answerability of Queries in Retrieval-Augmented Code Generation","date":"2024-11-08","arxiv_id":"2411.05547","repositories_listed":0,"syntology":null},{"url":null,"slug":"codelutra-boosting-llm-code-generation-via","title":"CodeLutra: Boosting LLM Code Generation via Preference-Guided Refinement","date":"2024-11-07","arxiv_id":"2411.05199","repositories_listed":0,"syntology":null},{"url":null,"slug":"codetree-agent-guided-tree-search-for-code","title":"CodeTree: Agent-guided Tree Search for Code Generation with Large Language Models","date":"2024-11-07","arxiv_id":"2411.04329","repositories_listed":0,"syntology":null},{"url":null,"slug":"opencoder-the-open-cookbook-for-top-tier-code","title":"OpenCoder: The Open Cookbook for Top-Tier Code Large Language Models","date":"2024-11-07","arxiv_id":"2411.04905","repositories_listed":0,"syntology":null},{"url":null,"slug":"crystal-illuminating-llm-abilities-on","title":"Crystal: Illuminating LLM Abilities on Language and Code","date":"2024-11-06","arxiv_id":"2411.04156","repositories_listed":0,"syntology":null},{"url":null,"slug":"defining-and-evaluating-physical-safety-for","title":"Defining and Evaluating Physical Safety for Large Language Models","date":"2024-11-04","arxiv_id":"2411.02317","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-table-representations-with-llm","title":"Enhancing Table Representations with LLM-powered Synthetic Data Generation","date":"2024-11-04","arxiv_id":"2411.03356","repositories_listed":0,"syntology":null},{"url":null,"slug":"eurekaverse-environment-curriculum-generation","title":"Eurekaverse: Environment Curriculum Generation via Large Language Models","date":"2024-11-04","arxiv_id":"2411.01775","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-ability-of-large-language-1","title":"Evaluating the Ability of Large Language Models to Generate Verifiable Specifications in VeriFast","date":"2024-11-04","arxiv_id":"2411.02318","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-fine-tuning-of-large-1","title":"Parameter-Efficient Fine-Tuning of Large Language Models for Unit Test Generation: An Empirical Study","date":"2024-11-04","arxiv_id":"2411.02462","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-dive-into-large-language-model-code","title":"A Deep Dive Into Large Language Model Code Generation Mistakes: What and Why?","date":"2024-11-03","arxiv_id":"2411.01414","repositories_listed":0,"syntology":null},{"url":null,"slug":"democraft-using-in-context-learning-to","title":"Demo-Craft: Using In-Context Learning to Improve Code Generation in Large Language Models","date":"2024-10-30","arxiv_id":"2411.00865","repositories_listed":0,"syntology":null},{"url":"/paper/evocodebench-an-evolving-code-generation-1","slug":"evocodebench-an-evolving-code-generation-1","title":"EvoCodeBench: An Evolving Code Generation Benchmark with Domain-Specific Evaluations","date":"2024-10-30","arxiv_id":"2410.22821","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evocodebench-an-evolving-code-generation-1#ran","syntology_url":"https://syntology.ai/paper/2410.22821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22821"}},"official":null}},{"url":null,"slug":"explainable-behavior-cloning-teaching-large","title":"Explainable Behavior Cloning: Teaching Large Language Model Agents through Learning by Demonstration","date":"2024-10-30","arxiv_id":"2410.22916","repositories_listed":0,"syntology":null},{"url":null,"slug":"visioncoder-empowering-multi-agent-auto","title":"MaCTG: Multi-Agent Collaborative Thought Graph for Automatic Programming","date":"2024-10-25","arxiv_id":"2410.19245","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-llm-agent-for-automatic-geospatial-data","title":"An LLM Agent for Automatic Geospatial Data Analysis","date":"2024-10-24","arxiv_id":"2410.18792","repositories_listed":0,"syntology":null},{"url":null,"slug":"watermarking-large-language-models-and-the","title":"Watermarking Large Language Models and the Generated Content: Opportunities and Challenges","date":"2024-10-24","arxiv_id":"2410.19096","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-language-models-with-better-multi","title":"Faster Language Models with Better Multi-Token Prediction Using Tensor Decomposition","date":"2024-10-23","arxiv_id":"2410.17765","repositories_listed":0,"syntology":null},{"url":null,"slug":"mojobench-language-modeling-and-benchmarks","title":"MojoBench: Language Modeling and Benchmarks for Mojo","date":"2024-10-23","arxiv_id":"2410.17736","repositories_listed":0,"syntology":null},{"url":null,"slug":"process-supervision-guided-policy","title":"Process Supervision-Guided Policy Optimization for Code Generation","date":"2024-10-23","arxiv_id":"2410.17621","repositories_listed":0,"syntology":null},{"url":null,"slug":"zip-fit-embedding-free-data-selection-via","title":"ZIP-FIT: Embedding-Free Data Selection via Compression-Based Alignment","date":"2024-10-23","arxiv_id":"2410.18194","repositories_listed":0,"syntology":null},{"url":null,"slug":"geocode-gpt-a-large-language-model-for","title":"GeoCode-GPT: A Large Language Model for Geospatial Code Generation Tasks","date":"2024-10-22","arxiv_id":"2410.17031","repositories_listed":0,"syntology":null},{"url":null,"slug":"scattered-forest-search-smarter-code-space","title":"Scattered Forest Search: Smarter Code Space Exploration with LLMs","date":"2024-10-22","arxiv_id":"2411.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-proof-generation-for-rust-code-via","title":"Automated Proof Generation for Rust Code via Self-Evolution","date":"2024-10-21","arxiv_id":"2410.15756","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-explained-keywords-empower-large","title":"Self-Explained Keywords Empower Large Language Models for Code Generation","date":"2024-10-21","arxiv_id":"2410.15966","repositories_listed":0,"syntology":null},{"url":null,"slug":"celi-controller-embedded-language-model","title":"CELI: Controller-Embedded Language Model Interactions","date":"2024-10-18","arxiv_id":"2410.14627","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-agents-for-code-generation-with","title":"Enhancing LLM Agents for Code Generation with Possibility and Pass-rate Prioritized Experience Replay","date":"2024-10-16","arxiv_id":"2410.12236","repositories_listed":0,"syntology":null},{"url":null,"slug":"evidence-of-cognitive-deficits","title":"Evidence of Cognitive Deficits andDevelopmental Advances in Generative AI: A Clock Drawing Test Analysis","date":"2024-10-15","arxiv_id":"2410.11756","repositories_listed":0,"syntology":null},{"url":null,"slug":"substance-beats-style-why-beginning-students","title":"Substance Beats Style: Why Beginning Students Fail to Code with LLMs","date":"2024-10-15","arxiv_id":"2410.19792","repositories_listed":0,"syntology":null},{"url":null,"slug":"flare-faithful-logic-aided-reasoning-and","title":"FLARE: Faithful Logic-Aided Reasoning and Exploration","date":"2024-10-14","arxiv_id":"2410.11900","repositories_listed":0,"syntology":null},{"url":null,"slug":"collu-bench-a-benchmark-for-predicting","title":"Collu-Bench: A Benchmark for Predicting Language Model Hallucinations in Code","date":"2024-10-13","arxiv_id":"2410.09997","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-driving-simulations-via","title":"Conversational Code Generation: a Case Study of Designing a Dialogue System for Generating Driving Scenarios for Testing Autonomous Vehicles","date":"2024-10-13","arxiv_id":"2410.09829","repositories_listed":0,"syntology":null},{"url":null,"slug":"impeding-llm-assisted-cheating-in","title":"Impeding LLM-assisted Cheating in Introductory Programming Assignments via Adversarial Perturbation","date":"2024-10-12","arxiv_id":"2410.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-driven-software-experimentation-with","title":"Test-driven Software Experimentation with LASSO: an LLM Prompt Benchmarking Example","date":"2024-10-11","arxiv_id":"2410.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-gender-bias-in-code-large-language","title":"Mitigating Gender Bias in Code Large Language Models via Model Editing","date":"2024-10-10","arxiv_id":"2410.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-makes-large-language-models-reason-in","title":"What Makes Large Language Models Reason in (Multi-Turn) Code Generation?","date":"2024-10-10","arxiv_id":"2410.08105","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-augmented-code-generation-using","title":"Context-Augmented Code Generation Using Programming Knowledge Graphs","date":"2024-10-09","arxiv_id":"2410.18251","repositories_listed":0,"syntology":null},{"url":"/paper/da-code-agent-data-science-code-generation","slug":"da-code-agent-data-science-code-generation","title":"DA-Code: Agent Data Science Code Generation Benchmark for Large Language Models","date":"2024-10-09","arxiv_id":"2410.07331","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/da-code-agent-data-science-code-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07331"}},"official":null}},{"url":null,"slug":"generating-cad-code-with-vision-language","title":"Generating CAD Code with Vision-Language Models for 3D Designs","date":"2024-10-07","arxiv_id":"2410.05340","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-the-loop-hyper-parameter-optimization-for","title":"In-the-loop Hyper-Parameter Optimization for LLM-Based Automated Design of Heuristics","date":"2024-10-07","arxiv_id":"2410.16309","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-how-hard-to-think-input-adaptive","title":"Learning How Hard to Think: Input-Adaptive Allocation of LM Computation","date":"2024-10-07","arxiv_id":"2410.04707","repositories_listed":0,"syntology":null}],"record_sha256":"c12f44f3b193e1e393465ddeb8628dac4268099f4a1e5a43493a1f92f97079d4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}