{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/9","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":17,"rows_per_page":100,"rows":[801,900],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/8","next":"/task/code-generation/papers/10","papers":[{"url":null,"slug":"rethinking-the-effects-of-data-contamination","title":"Rethinking the effects of data contamination in Code Intelligence","date":"2025-06-03","arxiv_id":"2506.02791","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow2code-evaluating-large-language-models","title":"Flow2Code: Evaluating Large Language Models for Flowchart-based Code Generation Capability","date":"2025-06-02","arxiv_id":"2506.02073","repositories_listed":0,"syntology":null},{"url":null,"slug":"researchcodebench-benchmarking-llms-on","title":"ResearchCodeBench: Benchmarking LLMs on Implementing Novel Machine Learning Research Code","date":"2025-06-02","arxiv_id":"2506.02314","repositories_listed":0,"syntology":null},{"url":null,"slug":"salad-systematic-assessment-of-machine","title":"SALAD: Systematic Assessment of Machine Unlearing on LLM-Aided Hardware Design","date":"2025-06-02","arxiv_id":"2506.02089","repositories_listed":0,"syntology":null},{"url":null,"slug":"legal-compliance-evaluation-of-smart","title":"Legal Compliance Evaluation of Smart Contracts Generated By Large Language Models","date":"2025-06-01","arxiv_id":"2506.00943","repositories_listed":0,"syntology":null},{"url":null,"slug":"coquir-a-comprehensive-benchmark-for-code","title":"CoQuIR: A Comprehensive Benchmark for Code Quality-Aware Information Retrieval","date":"2025-05-31","arxiv_id":"2506.11066","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reward-driven-automated-webshell-malicious","title":"A Reward-driven Automated Webshell Malicious-code Generator for Red-teaming","date":"2025-05-30","arxiv_id":"2505.24252","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascading-adversarial-bias-from-injection-to","title":"Cascading Adversarial Bias from Injection to Distillation in Language Models","date":"2025-05-30","arxiv_id":"2505.24842","repositories_listed":0,"syntology":null},{"url":null,"slug":"eye-of-judgement-dissecting-the-evaluation-of","title":"Eye of Judgement: Dissecting the Evaluation of Russian-speaking LLMs with POLLUX","date":"2025-05-30","arxiv_id":"2505.24616","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardtests-synthesizing-high-quality-test","title":"HardTests: Synthesizing High-Quality Test Cases for LLM Coding","date":"2025-05-30","arxiv_id":"2505.24098","repositories_listed":0,"syntology":null},{"url":null,"slug":"swifteval-developing-a-language-specific","title":"SwiftEval: Developing a Language-Specific Benchmark for LLM-generated Code Evaluation","date":"2025-05-30","arxiv_id":"2505.24324","repositories_listed":0,"syntology":null},{"url":null,"slug":"writing-zero-bridge-the-gap-between-non","title":"Writing-Zero: Bridge the Gap Between Non-verifiable Tasks and Verifiable Rewards","date":"2025-05-30","arxiv_id":"2506.00103","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-based-code-generation-with","title":"Enhancing LLM-Based Code Generation with Complexity Metrics: A Feedback-Driven Approach","date":"2025-05-29","arxiv_id":"2505.23953","repositories_listed":0,"syntology":null},{"url":null,"slug":"infinite-instruct-synthesizing-scaling-code","title":"Infinite-Instruct: Synthesizing Scaling Code instruction Data with Bidirectional Synthesis and Static Verification","date":"2025-05-29","arxiv_id":"2505.23177","repositories_listed":0,"syntology":null},{"url":null,"slug":"swingarena-competitive-programming-arena-for","title":"SwingArena: Competitive Programming Arena for Long-context GitHub Issue Solving","date":"2025-05-29","arxiv_id":"2505.23932","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeprtl2-a-versatile-model-for-rtl-related","title":"DeepRTL2: A Versatile Model for RTL-Related Tasks","date":"2025-05-28","arxiv_id":"2506.15697","repositories_listed":0,"syntology":null},{"url":null,"slug":"hilde-intentional-code-generation-via-human","title":"HiLDe: Intentional Code Generation via Human-in-the-Loop Decoding","date":"2025-05-28","arxiv_id":"2505.22906","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-llm-as-judge-metric-for-bridging-the-gap","title":"An LLM-as-Judge Metric for Bridging the Gap with Human Evaluation in SE Tasks","date":"2025-05-27","arxiv_id":"2505.20854","repositories_listed":0,"syntology":null},{"url":null,"slug":"rendering-aware-reinforcement-learning-for","title":"Rendering-Aware Reinforcement Learning for Vector Graphics Generation","date":"2025-05-27","arxiv_id":"2505.20793","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-diting-a-reasoning-based-metric-for","title":"CODE-DITING: A Reasoning-Based Metric for Functional Alignment in Code Evaluation","date":"2025-05-26","arxiv_id":"2505.19502","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-it-automation-tasks","title":"Large Language Models for IT Automation Tasks: Are We There Yet?","date":"2025-05-26","arxiv_id":"2505.20505","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-in-code-co-generation","title":"Large Language Models in Code Co-generation for Safe Autonomous Vehicles","date":"2025-05-26","arxiv_id":"2505.19658","repositories_listed":0,"syntology":null},{"url":null,"slug":"activedpo-active-direct-preference","title":"ActiveDPO: Active Direct Preference Optimization for Sample-Efficient Alignment","date":"2025-05-25","arxiv_id":"2505.19241","repositories_listed":0,"syntology":null},{"url":null,"slug":"architectures-of-error-a-philosophical","title":"Architectures of Error: A Philosophical Inquiry into AI and Human Code Generation","date":"2025-05-25","arxiv_id":"2505.19353","repositories_listed":0,"syntology":null},{"url":null,"slug":"autocomp-llm-driven-code-optimization-for","title":"Autocomp: LLM-Driven Code Optimization for Tensor Accelerators","date":"2025-05-24","arxiv_id":"2505.18574","repositories_listed":0,"syntology":null},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":null,"slug":"from-output-to-evaluation-does-raw","title":"From Output to Evaluation: Does Raw Instruction-Tuned Code LLMs Output Suffice for Fill-in-the-Middle Code Generation?","date":"2025-05-24","arxiv_id":"2505.18789","repositories_listed":0,"syntology":null},{"url":null,"slug":"hd-pissa-high-rank-distributed-orthogonal","title":"HD-PiSSA: High-Rank Distributed Orthogonal Adaptation","date":"2025-05-24","arxiv_id":"2505.18777","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptwise-online-learning-for-cost-aware","title":"PromptWise: Online Learning for Cost-Aware Prompt Assignment in Generative Models","date":"2025-05-24","arxiv_id":"2505.18901","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-energy-efficiency-of-the-code","title":"Evaluating the Energy-Efficiency of the Code Generated by LLMs","date":"2025-05-23","arxiv_id":"2505.20324","repositories_listed":0,"syntology":null},{"url":null,"slug":"ppt-a-process-based-preference-learning","title":"PPT: A Process-based Preference Learning Framework for Self Improving Table Question Answering Models","date":"2025-05-23","arxiv_id":"2505.17565","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-graph-model-cgm-a-graph-integrated-large","title":"Code Graph Model (CGM): A Graph-Integrated Large Language Model for Repository-Level Software Engineering Tasks","date":"2025-05-22","arxiv_id":"2505.16901","repositories_listed":0,"syntology":null},{"url":null,"slug":"maps-a-multilingual-benchmark-for-global","title":"MAPS: A Multilingual Benchmark for Global Agent Performance and Security","date":"2025-05-21","arxiv_id":"2505.15935","repositories_listed":0,"syntology":null},{"url":null,"slug":"simcopilot-evaluating-large-language-models","title":"SIMCOPILOT: Evaluating Large Language Models for Copilot-Style Code Generation","date":"2025-05-21","arxiv_id":"2505.21514","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-science-of-causal-interpretability","title":"Towards a Science of Causal Interpretability in Deep Learning for Software Engineering","date":"2025-05-21","arxiv_id":"2505.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheaper-better-faster-stronger-robust-text-to","title":"Cheaper, Better, Faster, Stronger: Robust Text-to-SQL without Chain-of-Thought or Fine-Tuning","date":"2025-05-20","arxiv_id":"2505.14174","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-reasoning-to-code-grpo-optimization-for","title":"From Reasoning to Code: GRPO Optimization for Underrepresented Languages","date":"2025-05-20","arxiv_id":"2506.11027","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-graph-based-repository-level-code","title":"Knowledge Graph Based Repository-Level Code Generation","date":"2025-05-20","arxiv_id":"2505.14394","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evolving-curriculum-for-llm-reasoning","title":"Self-Evolving Curriculum for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14970","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-beyond-discrete-token","title":"Text Generation Beyond Discrete Token Sampling","date":"2025-05-20","arxiv_id":"2505.14827","repositories_listed":0,"syntology":null},{"url":null,"slug":"autogeeval-a-multimodal-and-automated","title":"AutoGEEval: A Multimodal and Automated Framework for Geospatial Code Generation on GEE with Large Language Models","date":"2025-05-19","arxiv_id":"2505.12900","repositories_listed":0,"syntology":null},{"url":"/paper/krikri-advancing-open-large-language-models","slug":"krikri-advancing-open-large-language-models","title":"Krikri: Advancing Open Large Language Models for Greek","date":"2025-05-19","arxiv_id":"2505.13772","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-policy-optimization-with-group-equivalent","title":"On-Policy Optimization with Group Equivalent Preference for Multi-Programming Language Understanding","date":"2025-05-19","arxiv_id":"2505.12723","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-code-generation-for-functional","title":"Selective Code Generation for Functional Guarantees","date":"2025-05-19","arxiv_id":"2505.13553","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-complexity-in-videoqa-via","title":"Understanding Complexity in VideoQA via Visual Program Generation","date":"2025-05-19","arxiv_id":"2505.13429","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaloop-assessing-llm-robustness-in","title":"EVALOOP: Assessing LLM Robustness in Programming from a Self-consistency Perspective","date":"2025-05-18","arxiv_id":"2505.12185","repositories_listed":0,"syntology":null},{"url":null,"slug":"socia-an-end-to-end-agentic-framework-for","title":"SOCIA: An End-to-End Agentic Framework for Automated Cyber-Physical-Social Simulator Generation","date":"2025-05-17","arxiv_id":"2505.12006","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10937","title":"Reasoning with OmniThought: A Large CoT Dataset with Verbosity and Cognitive Difficulty Annotations","date":"2025-05-16","arxiv_id":"2505.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10749","title":"Code-Driven Planning in Grid Worlds with Large Language Models","date":"2025-05-15","arxiv_id":"2505.10749","repositories_listed":0,"syntology":null},{"url":null,"slug":"crpe-expanding-the-reasoning-capability-of","title":"CRPE: Expanding The Reasoning Capability of Large Language Model for Code Generation","date":"2025-05-15","arxiv_id":"2505.10594","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-the-diffusion-chain-of-lateral","title":"Reinforcing the Diffusion Chain of Lateral Thought with Diffusion Language Models","date":"2025-05-15","arxiv_id":"2505.10446","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-as-a-service-based-on-agent-network","title":"Agent-as-a-Service based on Agent Network","date":"2025-05-13","arxiv_id":"2505.08446","repositories_listed":0,"syntology":null},{"url":null,"slug":"cad-coder-text-guided-cad-files-code","title":"CAD-Coder:Text-Guided CAD Files Code Generation","date":"2025-05-13","arxiv_id":"2505.08686","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-llm-metrics-through-real-world","title":"Evaluating LLM Metrics Through Real-World Capabilities","date":"2025-05-13","arxiv_id":"2505.08253","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-large-language-model-usability","title":"Generalizing Large Language Model Usability Across Resource-Constrained","date":"2025-05-13","arxiv_id":"2505.17040","repositories_listed":0,"syntology":null},{"url":null,"slug":"tests-as-prompt-a-test-driven-development","title":"Tests as Prompt: A Test-Driven-Development Benchmark for LLM Code Generation","date":"2025-05-13","arxiv_id":"2505.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-trigger-token-is-enough-a-defense","title":"One Trigger Token Is Enough: A Defense Strategy for Balancing Safety and Usability in Large Language Models","date":"2025-05-12","arxiv_id":"2505.07167","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-retrieval-for-milp-instance-generation","title":"Code Retrieval for MILP Instance Generation","date":"2025-05-11","arxiv_id":"2505.11526","repositories_listed":0,"syntology":null},{"url":null,"slug":"rtl-graph-enhanced-llm-for-rtl-code","title":"RTL++: Graph-enhanced LLM for RTL Code Generation","date":"2025-05-11","arxiv_id":"2505.13479","repositories_listed":0,"syntology":null},{"url":null,"slug":"codemixbench-evaluating-large-language-models","title":"CodeMixBench: Evaluating Large Language Models on Code Generation with Code-Mixed Prompts","date":"2025-05-08","arxiv_id":"2505.05063","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-proposal-for-evaluating-the-operational","title":"A Proposal for Evaluating the Operational Risk for ChatBots based on Large Language Models","date":"2025-05-07","arxiv_id":"2505.04784","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-code-customization-with-visual-results-a","title":"LLM Code Customization with Visual Results: A Benchmark on TikZ","date":"2025-05-07","arxiv_id":"2505.04670","repositories_listed":0,"syntology":null},{"url":null,"slug":"yabloco-yet-another-benchmark-for-long","title":"YABLoCo: Yet Another Benchmark for Long Context Code Generation","date":"2025-05-07","arxiv_id":"2505.04406","repositories_listed":0,"syntology":null},{"url":null,"slug":"capability-driven-skill-generation-with-llms","title":"Capability-Driven Skill Generation with LLMs: A RAG-Based Approach for Reusing Existing Libraries and Interfaces","date":"2025-05-06","arxiv_id":"2505.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"marco-a-multi-agent-system-for-optimizing-hpc","title":"MARCO: Multi-Agent Code Optimization with Real-Time Knowledge Integration for High-Performance Computing","date":"2025-05-06","arxiv_id":"2505.03906","repositories_listed":0,"syntology":null},{"url":null,"slug":"scratch-copilot-supporting-youth-creative","title":"Scratch Copilot: Supporting Youth Creative Coding with AI","date":"2025-05-06","arxiv_id":"2505.03867","repositories_listed":0,"syntology":null},{"url":null,"slug":"story2game-generating-almost-everything-in-an","title":"STORY2GAME: Generating (Almost) Everything in an Interactive Fiction Game","date":"2025-05-06","arxiv_id":"2505.03547","repositories_listed":0,"syntology":null},{"url":null,"slug":"akd-adversarial-knowledge-distillation-for","title":"AKD : Adversarial Knowledge Distillation For Large Language Models Alignment on Coding tasks","date":"2025-05-05","arxiv_id":"2505.06267","repositories_listed":0,"syntology":null},{"url":null,"slug":"qimeng-xpiler-transcompiling-tensor-programs","title":"QiMeng-Xpiler: Transcompiling Tensor Programs for Deep Learning Systems with a Neural-Symbolic Approach","date":"2025-05-04","arxiv_id":"2505.02146","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-rusty-link-in-the-ai-supply-chain-detecting","title":"A Rusty Link in the AI Supply Chain: Detecting Evil Configurations in Model Repositories","date":"2025-05-02","arxiv_id":"2505.01067","repositories_listed":0,"syntology":null},{"url":null,"slug":"chorus-zero-shot-hierarchical-retrieval-and","title":"CHORUS: Zero-shot Hierarchical Retrieval and Orchestration for Generating Linear Programming Code","date":"2025-05-02","arxiv_id":"2505.01485","repositories_listed":0,"syntology":null},{"url":null,"slug":"pipespec-breaking-stage-dependencies-in","title":"PipeSpec: Breaking Stage Dependencies in Hierarchical LLM Decoding","date":"2025-05-02","arxiv_id":"2505.01572","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-llm-code-generation-quality-through","title":"Assessing LLM code generation quality through path planning tasks","date":"2025-04-30","arxiv_id":"2504.21276","repositories_listed":0,"syntology":null},{"url":null,"slug":"arcs-agentic-retrieval-augmented-code","title":"ARCS: Agentic Retrieval-Augmented Code Synthesis with Iterative Refinement","date":"2025-04-29","arxiv_id":"2504.20434","repositories_listed":0,"syntology":null},{"url":null,"slug":"coco-bench-a-comprehensive-code-benchmark-for","title":"CoCo-Bench: A Comprehensive Code Benchmark For Multi-task Large Language Model Evaluation","date":"2025-04-29","arxiv_id":"2504.20673","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-by-code-generation-llms","title":"Hallucination by Code Generation LLMs: Taxonomy, Benchmarks, Mitigation, and Challenges","date":"2025-04-29","arxiv_id":"2504.20799","repositories_listed":0,"syntology":null},{"url":null,"slug":"secrepobench-benchmarking-llms-for-secure","title":"SecRepoBench: Benchmarking LLMs for Secure Code Generation in Real-World Repositories","date":"2025-04-29","arxiv_id":"2504.21205","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-discovery-for-software-scripting","title":"Skill Discovery for Software Scripting Automation via Offline Simulations with LLMs","date":"2025-04-29","arxiv_id":"2504.20406","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hidden-risks-of-llm-generated-web","title":"The Hidden Risks of LLM-Generated Web Application Code: A Security-Centric Evaluation of Code Generation Capabilities in Large Language Models","date":"2025-04-29","arxiv_id":"2504.20612","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-automated-reinforcement-learning-reward","title":"An Automated Reinforcement Learning Reward Design Framework with Large Language Model for Cooperative Platoon Coordination","date":"2025-04-28","arxiv_id":"2504.19480","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-grounded-reasoning-by-code","title":"Evaluating Grounded Reasoning by Code-Assisted Large Language Models for Mathematics","date":"2025-04-24","arxiv_id":"2504.17665","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-and-complex-test-data","title":"High-Fidelity And Complex Test Data Generation For Real-World SQL Code Generation Services","date":"2025-04-24","arxiv_id":"2504.17203","repositories_listed":0,"syntology":null},{"url":null,"slug":"clarifycoder-clarification-aware-fine-tuning","title":"ClarifyCoder: Clarification-Aware Fine-Tuning for Programmatic Problem Solving","date":"2025-04-23","arxiv_id":"2504.16331","repositories_listed":0,"syntology":null},{"url":null,"slug":"edubot-can-llms-solve-personalized-learning","title":"EduBot -- Can LLMs Solve Personalized Learning and Programming Assignments?","date":"2025-04-23","arxiv_id":"2504.17824","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-class-level-benchmark-dataset","title":"A Large-scale Class-level Benchmark Dataset for Code Generation with LLMs","date":"2025-04-22","arxiv_id":"2504.15564","repositories_listed":0,"syntology":null},{"url":null,"slug":"insights-from-verification-training-a-verilog","title":"Insights from Verification: Training a Verilog Generation LLM with Reinforcement Learning with Testbench Feedback","date":"2025-04-22","arxiv_id":"2504.15804","repositories_listed":0,"syntology":null},{"url":null,"slug":"vericoder-enhancing-llm-based-rtl-code","title":"VeriCoder: Enhancing LLM-Based RTL Code Generation through Functional Correctness Validation","date":"2025-04-22","arxiv_id":"2504.15659","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-ai-to-generate-better-ai-code","title":"Empowering AI to Generate Better AI Code: Guided Generation of Deep Learning Projects with LLMs","date":"2025-04-21","arxiv_id":"2504.15080","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-code-generation-of-llms-in","title":"Evaluating Code Generation of LLMs in Advanced Computer Science Problems","date":"2025-04-21","arxiv_id":"2504.14964","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rl-exploration-for-llm-reasoning","title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","date":"2025-04-19","arxiv_id":"2504.14363","repositories_listed":0,"syntology":null},{"url":null,"slug":"codevisionary-an-agent-based-framework-for","title":"CodeVisionary: An Agent-based Framework for Evaluating Large Language Models in Code Generation","date":"2025-04-18","arxiv_id":"2504.13472","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-prompt-patterns-affect-code-quality-a","title":"Do Prompt Patterns Affect Code Quality? A First Empirical Assessment of ChatGPT-Generated Code","date":"2025-04-18","arxiv_id":"2504.13656","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-end-to-end-network-intent-management","title":"Towards End-to-End Network Intent Management with Large Language Models","date":"2025-04-18","arxiv_id":"2504.13589","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-copycat-conundrum-demystifying","title":"Code Copycat Conundrum: Demystifying Repetition in LLM-based Code Generation","date":"2025-04-17","arxiv_id":"2504.12608","repositories_listed":0,"syntology":null},{"url":null,"slug":"syntactic-and-semantic-control-of-large","title":"Syntactic and Semantic Control of Large Language Models via Sequential Monte Carlo","date":"2025-04-17","arxiv_id":"2504.13139","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-generation-of-high-quality-cot","title":"Rethinking the Generation of High-Quality CoT Data from the Perspective of LLM-Adaptive Question Difficulty Grading","date":"2025-04-16","arxiv_id":"2504.11919","repositories_listed":0,"syntology":null},{"url":null,"slug":"themisto-jupyter-based-runtime-benchmark","title":"Themisto: Jupyter-Based Runtime Benchmark","date":"2025-04-16","arxiv_id":"2504.12365","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-future-of-mllm-prompting-is-adaptive-a","title":"The Future of MLLM Prompting is Adaptive: A Comprehensive Experimental Evaluation of Prompt Engineering Methods for Robust Multimodal Performance","date":"2025-04-14","arxiv_id":"2504.10179","repositories_listed":0,"syntology":null},{"url":null,"slug":"draw-with-thought-unleashing-multimodal","title":"Draw with Thought: Unleashing Multimodal Reasoning for Scientific Diagram Generation","date":"2025-04-13","arxiv_id":"2504.09479","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-self-training-for-code-generation","title":"Iterative Self-Training for Code Generation via Reinforced Re-Ranking","date":"2025-04-13","arxiv_id":"2504.09643","repositories_listed":0,"syntology":null}],"record_sha256":"86dce876770d916760ce4e9d392c5ec5b143042ca97cf4f1352287996b1bfcd1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}