{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/4","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":17,"rows_per_page":100,"rows":[301,400],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/3","next":"/task/code-generation/papers/5","papers":[{"url":"/paper/cornstack-high-quality-contrastive-data-for","slug":"cornstack-high-quality-contrastive-data-for","title":"CoRNStack: High-Quality Contrastive Data for Better Code Retrieval and Reranking","date":"2024-12-01","arxiv_id":"2412.01007","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cornstack-high-quality-contrastive-data-for#ran","syntology_url":"https://syntology.ai/paper/2412.01007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01007"}},"official":{"repos":["gangiswag/cornstack"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scratcheval-are-gpt-4o-smarter-than-my-child","slug":"scratcheval-are-gpt-4o-smarter-than-my-child","title":"ScratchEval: Are GPT-4o Smarter than My Child? Evaluating Large Multimodal Models with Visual Programming Challenges","date":"2024-11-28","arxiv_id":"2411.18932","repositories_listed":1,"syntology":null},{"url":"/paper/using-a-feedback-loop-for-llm-based","slug":"using-a-feedback-loop-for-llm-based","title":"Using a Feedback Loop for LLM-based Infrastructure as Code Generation","date":"2024-11-28","arxiv_id":"2411.19043","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-brained-gui-agents-a","slug":"large-language-model-brained-gui-agents-a","title":"Large Language Model-Brained GUI Agents: A Survey","date":"2024-11-27","arxiv_id":"2411.18279","repositories_listed":1,"syntology":null},{"url":"/paper/words-matter-leveraging-individual-text","slug":"words-matter-leveraging-individual-text","title":"Words Matter: Leveraging Individual Text Embeddings for Code Generation in CLIP Test-Time Adaptation","date":"2024-11-26","arxiv_id":"2411.17002","repositories_listed":1,"syntology":null},{"url":"/paper/instruct-or-interact-exploring-and-eliciting","slug":"instruct-or-interact-exploring-and-eliciting","title":"Instruct or Interact? Exploring and Eliciting LLMs' Capability in Code Snippet Adaptation Through Prompt Engineering","date":"2024-11-23","arxiv_id":"2411.15501","repositories_listed":1,"syntology":null},{"url":"/paper/planning-driven-programming-a-large-language","slug":"planning-driven-programming-a-large-language","title":"Planning-Driven Programming: A Large Language Model Programming Workflow","date":"2024-11-21","arxiv_id":"2411.14503","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-driven-programming-a-large-language#ran","syntology_url":"https://syntology.ai/paper/2411.14503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14503"}},"official":{"repos":["you68681/lpw"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prosec-fortifying-code-llms-with-proactive","slug":"prosec-fortifying-code-llms-with-proactive","title":"ProSec: Fortifying Code LLMs with Proactive Security Alignment","date":"2024-11-19","arxiv_id":"2411.12882","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prosec-fortifying-code-llms-with-proactive#ran","syntology_url":"https://syntology.ai/paper/2411.12882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12882"}},"official":{"repos":["PurCL/ProSec"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sra-mcts-self-driven-reasoning-aurmentation","slug":"sra-mcts-self-driven-reasoning-aurmentation","title":"SRA-MCTS: Self-driven Reasoning Augmentation with Monte Carlo Tree Search for Code Generation","date":"2024-11-17","arxiv_id":"2411.11053","repositories_listed":1,"syntology":null},{"url":"/paper/squeezed-attention-accelerating-long-context","slug":"squeezed-attention-accelerating-long-context","title":"Squeezed Attention: Accelerating Long Context Length LLM Inference","date":"2024-11-14","arxiv_id":"2411.09688","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/squeezed-attention-accelerating-long-context#ran","syntology_url":"https://syntology.ai/paper/2411.09688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.09688"}},"official":{"repos":["SqueezeAILab/SqueezedAttention"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pygen-a-collaborative-human-ai-approach-to","slug":"pygen-a-collaborative-human-ai-approach-to","title":"PyGen: A Collaborative Human-AI Approach to Python Package Creation","date":"2024-11-13","arxiv_id":"2411.08932","repositories_listed":1,"syntology":null},{"url":"/paper/model-editing-for-llms4code-how-far-are-we","slug":"model-editing-for-llms4code-how-far-are-we","title":"Model Editing for LLMs4Code: How Far are We?","date":"2024-11-11","arxiv_id":"2411.06638","repositories_listed":1,"syntology":null},{"url":"/paper/utmath-math-evaluation-with-unit-test-via","slug":"utmath-math-evaluation-with-unit-test-via","title":"UTMath: Math Evaluation with Unit Test via Reasoning-to-Coding Thoughts","date":"2024-11-11","arxiv_id":"2411.07240","repositories_listed":1,"syntology":null},{"url":"/paper/alignxie-improving-multilingual-information","slug":"alignxie-improving-multilingual-information","title":"AlignXIE: Improving Multilingual Information Extraction by Cross-Lingual Alignment","date":"2024-11-07","arxiv_id":"2411.04794","repositories_listed":1,"syntology":null},{"url":"/paper/suffixdecoding-a-model-free-approach-to","slug":"suffixdecoding-a-model-free-approach-to","title":"SuffixDecoding: Extreme Speculative Decoding for Emerging AI Applications","date":"2024-11-07","arxiv_id":"2411.04975","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/suffixdecoding-a-model-free-approach-to#ran","syntology_url":"https://syntology.ai/paper/2411.04975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04975"}},"official":{"repos":["snowflakedb/arcticinference"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gis-copilot-towards-an-autonomous-gis-agent","slug":"gis-copilot-towards-an-autonomous-gis-agent","title":"GIS Copilot: Towards an Autonomous GIS Agent for Spatial Analysis","date":"2024-11-05","arxiv_id":"2411.03205","repositories_listed":1,"syntology":null},{"url":"/paper/gitchameleon-unmasking-the-version-switching","slug":"gitchameleon-unmasking-the-version-switching","title":"GitChameleon: Unmasking the Version-Switching Capabilities of Code Generation Models","date":"2024-11-05","arxiv_id":"2411.05830","repositories_listed":1,"syntology":null},{"url":"/paper/interaction2code-how-far-are-we-from","slug":"interaction2code-how-far-are-we-from","title":"Interaction2Code: Benchmarking MLLM-based Interactive Webpage Code Generation from Interactive Prototyping","date":"2024-11-05","arxiv_id":"2411.03292","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interaction2code-how-far-are-we-from#ran","syntology_url":"https://syntology.ai/paper/2411.03292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03292"}},"official":{"repos":["webpai/interaction2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metrex-a-benchmark-for-verilog-code-metric","slug":"metrex-a-benchmark-for-verilog-code-metric","title":"MetRex: A Benchmark for Verilog Code Metric Reasoning Using LLMs","date":"2024-11-05","arxiv_id":"2411.03471","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/metrex-a-benchmark-for-verilog-code-metric#ran","syntology_url":"https://syntology.ai/paper/2411.03471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03471"}},"official":{"repos":["scale-lab/MetRex"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/autovfx-physically-realistic-video-editing","slug":"autovfx-physically-realistic-video-editing","title":"AutoVFX: Physically Realistic Video Editing from Natural Language Instructions","date":"2024-11-04","arxiv_id":"2411.02394","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/autovfx-physically-realistic-video-editing#ran","syntology_url":"https://syntology.ai/paper/2411.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02394"}},"official":null}},{"url":"/paper/neo-saving-gpu-memory-crisis-with-cpu","slug":"neo-saving-gpu-memory-crisis-with-cpu","title":"NEO: Saving GPU Memory Crisis with CPU Offloading for Online LLM Inference","date":"2024-11-02","arxiv_id":"2411.01142","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/neo-saving-gpu-memory-crisis-with-cpu#ran","syntology_url":"https://syntology.ai/paper/2411.01142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01142"}},"official":null}},{"url":"/paper/turtlebench-a-visual-programming-benchmark-in","slug":"turtlebench-a-visual-programming-benchmark-in","title":"TurtleBench: A Visual Programming Benchmark in Turtle Geometry","date":"2024-10-31","arxiv_id":"2411.00264","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turtlebench-a-visual-programming-benchmark-in#ran","syntology_url":"https://syntology.ai/paper/2411.00264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00264"}},"official":{"repos":["sinaris76/turtlebench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-language-models-replace-programmers","slug":"can-language-models-replace-programmers","title":"Can Language Models Replace Programmers for Coding? REPOCOD Says 'Not Yet'","date":"2024-10-29","arxiv_id":"2410.21647","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-llm-inference-with-optimized-sample","slug":"scaling-llm-inference-with-optimized-sample","title":"Scaling LLM Inference with Optimized Sample Compute Allocation","date":"2024-10-29","arxiv_id":"2410.22480","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-llm-inference-with-optimized-sample#ran","syntology_url":"https://syntology.ai/paper/2410.22480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22480"}},"official":{"repos":["leililab/osca"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-autoregression-fast-llms-via-self","slug":"beyond-autoregression-fast-llms-via-self","title":"Beyond Autoregression: Fast LLMs via Self-Distillation Through Time","date":"2024-10-28","arxiv_id":"2410.21035","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-autoregression-fast-llms-via-self#ran","syntology_url":"https://syntology.ai/paper/2410.21035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21035"}},"official":{"repos":["jdeschena/sdtt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/falcon-feedback-driven-adaptive-long-short","slug":"falcon-feedback-driven-adaptive-long-short","title":"FALCON: Feedback-driven Adaptive Long/short-term memory reinforced Coding Optimization system","date":"2024-10-28","arxiv_id":"2410.21349","repositories_listed":1,"syntology":null},{"url":"/paper/geo-fub-a-method-for-constructing-an-operator","slug":"geo-fub-a-method-for-constructing-an-operator","title":"Geo-FuB: A Method for Constructing an Operator-Function Knowledge Base for Geospatial Code Generation Tasks Using Large Language Models","date":"2024-10-28","arxiv_id":"2410.20975","repositories_listed":1,"syntology":null},{"url":"/paper/spicepilot-navigating-spice-code-generation","slug":"spicepilot-navigating-spice-code-generation","title":"SPICEPilot: Navigating SPICE Code Generation and Simulation with AI Guidance","date":"2024-10-27","arxiv_id":"2410.20553","repositories_listed":1,"syntology":null},{"url":"/paper/waffle-multi-modal-model-for-automated-front","slug":"waffle-multi-modal-model-for-automated-front","title":"WAFFLE: Finetuning Multi-Modal Model for Automated Front-End Development","date":"2024-10-24","arxiv_id":"2410.18362","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-coding-assistant-via-the-retrieval","slug":"building-a-coding-assistant-via-the-retrieval","title":"Building A Coding Assistant via the Retrieval-Augmented Language Model","date":"2024-10-21","arxiv_id":"2410.16229","repositories_listed":1,"syntology":null},{"url":"/paper/improving-parallel-program-performance","slug":"improving-parallel-program-performance","title":"Improving Parallel Program Performance with LLM Optimizers via Agent-System Interfaces","date":"2024-10-21","arxiv_id":"2410.15625","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-in-computer-science","slug":"large-language-models-in-computer-science","title":"Large Language Models in Computer Science Education: A Systematic Literature Review","date":"2024-10-21","arxiv_id":"2410.16349","repositories_listed":1,"syntology":null},{"url":"/paper/mccoder-streamlining-motion-control-with-llm","slug":"mccoder-streamlining-motion-control-with-llm","title":"MCCoder: Streamlining Motion Control with LLM-Assisted Code Generation and Rigorous Verification","date":"2024-10-19","arxiv_id":"2410.15154","repositories_listed":1,"syntology":null},{"url":"/paper/mhumaneval-a-multilingual-benchmark-to","slug":"mhumaneval-a-multilingual-benchmark-to","title":"mHumanEval -- A Multilingual Benchmark to Evaluate Large Language Models for Code Generation","date":"2024-10-19","arxiv_id":"2410.15037","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-quantized-large-language-models-1","slug":"evaluating-quantized-large-language-models-1","title":"Evaluating Quantized Large Language Models for Code Generation on Low-Resource Language Benchmarks","date":"2024-10-18","arxiv_id":"2410.14766","repositories_listed":1,"syntology":null},{"url":"/paper/llmopt-learning-to-define-and-solve-general","slug":"llmopt-learning-to-define-and-solve-general","title":"LLMOPT: Learning to Define and Solve General Optimization Problems from Scratch","date":"2024-10-17","arxiv_id":"2410.13213","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llmopt-learning-to-define-and-solve-general#ran","syntology_url":"https://syntology.ai/paper/2410.13213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13213"}},"official":{"repos":["caigaojiang/llmopt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-v-evaluating-visual-understanding","slug":"humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","arxiv_id":"2410.12381","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-v-evaluating-visual-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.12381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12381"}},"official":{"repos":["HumanEval-V/HumanEval-V-Benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/personality-guided-code-generation-using","slug":"personality-guided-code-generation-using","title":"Personality-Guided Code Generation Using Large Language Models","date":"2024-10-16","arxiv_id":"2411.00006","repositories_listed":1,"syntology":null},{"url":"/paper/fveval-understanding-language-model","slug":"fveval-understanding-language-model","title":"FVEval: Understanding Language Model Capabilities in Formal Verification of Digital Hardware","date":"2024-10-15","arxiv_id":"2410.23299","repositories_listed":1,"syntology":null},{"url":"/paper/agent-as-a-judge-evaluate-agents-with-agents","slug":"agent-as-a-judge-evaluate-agents-with-agents","title":"Agent-as-a-Judge: Evaluate Agents with Agents","date":"2024-10-14","arxiv_id":"2410.10934","repositories_listed":1,"syntology":null},{"url":"/paper/coral-order-agnostic-language-modeling-for","slug":"coral-order-agnostic-language-modeling-for","title":"COrAL: Order-Agnostic Language Modeling for Efficient Iterative Refinement","date":"2024-10-12","arxiv_id":"2410.09675","repositories_listed":1,"syntology":null},{"url":"/paper/pear-a-robust-and-flexible-automation","slug":"pear-a-robust-and-flexible-automation","title":"PEAR: A Robust and Flexible Automation Framework for Ptychography Enabled by Multiple Large Language Model Agents","date":"2024-10-11","arxiv_id":"2410.09034","repositories_listed":1,"syntology":null},{"url":"/paper/software-engineering-and-foundation-models","slug":"software-engineering-and-foundation-models","title":"Software Engineering and Foundation Models: Insights from Industry Blogs Using a Jury of Foundation Models","date":"2024-10-11","arxiv_id":"2410.09012","repositories_listed":1,"syntology":null},{"url":"/paper/itergen-iterative-structured-llm-generation","slug":"itergen-iterative-structured-llm-generation","title":"IterGen: Iterative Semantic-aware Structured LLM Generation with Backtracking","date":"2024-10-09","arxiv_id":"2410.07295","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/itergen-iterative-structured-llm-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07295"}},"official":{"repos":["uiuc-arc/itergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluation-of-code-llms-on-geospatial-code","slug":"evaluation-of-code-llms-on-geospatial-code","title":"Evaluation of Code LLMs on Geospatial Code Generation","date":"2024-10-06","arxiv_id":"2410.04617","repositories_listed":1,"syntology":null},{"url":"/paper/learning-code-preference-via-synthetic","slug":"learning-code-preference-via-synthetic","title":"Learning Code Preference via Synthetic Evolution","date":"2024-10-04","arxiv_id":"2410.03837","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-code-preference-via-synthetic#ran","syntology_url":"https://syntology.ai/paper/2410.03837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03837"}},"official":{"repos":["amazon-science/llm-code-preference"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-large-language-models-between-code","slug":"steering-large-language-models-between-code","title":"Steering Large Language Models between Code Execution and Textual Reasoning","date":"2024-10-04","arxiv_id":"2410.03524","repositories_listed":1,"syntology":null},{"url":"/paper/tadashi-enabling-ai-based-automated-code","slug":"tadashi-enabling-ai-based-automated-code","title":"Tadashi: Enabling AI-Based Automated Code Generation With Guaranteed Correctness","date":"2024-10-04","arxiv_id":"2410.03210","repositories_listed":1,"syntology":null},{"url":"/paper/automl-agent-a-multi-agent-llm-framework-for","slug":"automl-agent-a-multi-agent-llm-framework-for","title":"AutoML-Agent: A Multi-Agent LLM Framework for Full-Pipeline AutoML","date":"2024-10-03","arxiv_id":"2410.02958","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automl-agent-a-multi-agent-llm-framework-for#ran","syntology_url":"https://syntology.ai/paper/2410.02958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02958"}},"official":{"repos":["DeepAuto-AI/automl-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codejudge-evaluating-code-generation-with","slug":"codejudge-evaluating-code-generation-with","title":"CodeJudge: Evaluating Code Generation with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02184","repositories_listed":1,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":8,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/codejudge-evaluating-code-generation-with#ran","syntology_url":"https://syntology.ai/paper/2410.02184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02184"}},"official":{"repos":["VichyTong/CodeJudge"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ma-rlhf-reinforcement-learning-from-human","slug":"ma-rlhf-reinforcement-learning-from-human","title":"MA-RLHF: Reinforcement Learning from Human Feedback with Macro Actions","date":"2024-10-03","arxiv_id":"2410.02743","repositories_listed":1,"syntology":null},{"url":"/paper/repograph-enhancing-ai-software-engineering","slug":"repograph-enhancing-ai-software-engineering","title":"RepoGraph: Enhancing AI Software Engineering with Repository-level Code Graph","date":"2024-10-03","arxiv_id":"2410.14684","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/repograph-enhancing-ai-software-engineering#ran","syntology_url":"https://syntology.ai/paper/2410.14684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14684"}},"official":{"repos":["ozyyshr/repograph"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codev-bench-how-do-llms-understand-developer","slug":"codev-bench-how-do-llms-understand-developer","title":"Codev-Bench: How Do LLMs Understand Developer-Centric Code Completion?","date":"2024-10-02","arxiv_id":"2410.01353","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codev-bench-how-do-llms-understand-developer#ran","syntology_url":"https://syntology.ai/paper/2410.01353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01353"}},"official":{"repos":["LingmaTongyi/Codev-Bench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-code-to-correctness-closing-the-last","slug":"from-code-to-correctness-closing-the-last","title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","date":"2024-10-02","arxiv_id":"2410.01215","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/from-code-to-correctness-closing-the-last#ran","syntology_url":"https://syntology.ai/paper/2410.01215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01215"}},"official":{"repos":["YerbaPage/MGDebugger"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rgd-multi-llm-based-agent-debugger-via","slug":"rgd-multi-llm-based-agent-debugger-via","title":"RGD: Multi-LLM Based Agent Debugger via Refinement and Generation Guidance","date":"2024-10-02","arxiv_id":"2410.01242","repositories_listed":1,"syntology":null},{"url":"/paper/amr-evol-adaptive-modular-response-evolution","slug":"amr-evol-adaptive-modular-response-evolution","title":"AMR-Evol: Adaptive Modular Response Evolution Elicits Better Knowledge Distillation for Large Language Models in Code Generation","date":"2024-10-01","arxiv_id":"2410.00558","repositories_listed":1,"syntology":null},{"url":"/paper/babelbench-an-omni-benchmark-for-code-driven","slug":"babelbench-an-omni-benchmark-for-code-driven","title":"BabelBench: An Omni Benchmark for Code-Driven Analysis of Multimodal and Multistructured Data","date":"2024-10-01","arxiv_id":"2410.00773","repositories_listed":1,"syntology":null},{"url":"/paper/dreamstruct-understanding-slides-and-user","slug":"dreamstruct-understanding-slides-and-user","title":"DreamStruct: Understanding Slides and User Interfaces via Synthetic Data Generation","date":"2024-09-30","arxiv_id":"2410.00201","repositories_listed":1,"syntology":null},{"url":"/paper/llm-hallucinations-in-practical-code","slug":"llm-hallucinations-in-practical-code","title":"LLM Hallucinations in Practical Code Generation: Phenomena, Mechanism, and Mitigation","date":"2024-09-30","arxiv_id":"2409.20550","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-hallucinations-in-practical-code#ran","syntology_url":"https://syntology.ai/paper/2409.20550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20550"}},"official":{"repos":["deepsoftwareanalytics/llmcodinghallucination"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codeinsight-a-curated-dataset-of-practical","slug":"codeinsight-a-curated-dataset-of-practical","title":"CodeInsight: A Curated Dataset of Practical Coding Solutions from Stack Overflow","date":"2024-09-25","arxiv_id":"2409.16819","repositories_listed":1,"syntology":null},{"url":"/paper/moss-enabling-code-driven-evolution-and","slug":"moss-enabling-code-driven-evolution-and","title":"MOSS: Enabling Code-Driven Evolution and Context Management for AI Agents","date":"2024-09-24","arxiv_id":"2409.16120","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moss-enabling-code-driven-evolution-and#ran","syntology_url":"https://syntology.ai/paper/2409.16120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16120"}},"official":{"repos":["ghost-in-moss/ghostos"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rambo-enhancing-rag-based-repository-level","slug":"rambo-enhancing-rag-based-repository-level","title":"RAMBO: Enhancing RAG-based Repository-Level Method Body Completion","date":"2024-09-23","arxiv_id":"2409.15204","repositories_listed":1,"syntology":null},{"url":"/paper/rmcbench-benchmarking-large-language-models","slug":"rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","arxiv_id":"2409.15154","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rmcbench-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2409.15154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15154"}},"official":{"repos":["qing-yuan233/RMCBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualized-data-wrangling-code-generation","slug":"contextualized-data-wrangling-code-generation","title":"Contextualized Data-Wrangling Code Generation in Computational Notebooks","date":"2024-09-20","arxiv_id":"2409.13551","repositories_listed":1,"syntology":null},{"url":"/paper/2409-12993","slug":"2409-12993","title":"CraftRTL: High-quality Synthetic Data Generation for Verilog Code Models with Correct-by-Construction Non-Textual Representations and Targeted Code Repair","date":"2024-09-19","arxiv_id":"2409.12993","repositories_listed":1,"syntology":null},{"url":"/paper/a-case-study-of-web-app-coding-with-openai","slug":"a-case-study-of-web-app-coding-with-openai","title":"A Case Study of Web App Coding with OpenAI Reasoning Models","date":"2024-09-19","arxiv_id":"2409.13773","repositories_listed":1,"syntology":null},{"url":"/paper/autoverus-automated-proof-generation-for-rust","slug":"autoverus-automated-proof-generation-for-rust","title":"AutoVerus: Automated Proof Generation for Rust Code","date":"2024-09-19","arxiv_id":"2409.13082","repositories_listed":1,"syntology":null},{"url":"/paper/promsec-prompt-optimization-for-secure","slug":"promsec-prompt-optimization-for-secure","title":"PromSec: Prompt Optimization for Secure Generation of Functional Source Code with Large Language Models (LLMs)","date":"2024-09-19","arxiv_id":"2409.12699","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promsec-prompt-optimization-for-secure#ran","syntology_url":"https://syntology.ai/paper/2409.12699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12699"}},"official":{"repos":["mahmoudkanazzal/PromSec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autosafecoder-a-multi-agent-framework-for","slug":"autosafecoder-a-multi-agent-framework-for","title":"AutoSafeCoder: A Multi-Agent Framework for Securing LLM Code Generation through Static Analysis and Fuzz Testing","date":"2024-09-16","arxiv_id":"2409.10737","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/autosafecoder-a-multi-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.10737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10737"}},"official":{"repos":["secureaiautonomylab/autosafecoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/b4-towards-optimal-assessment-of-plausible","slug":"b4-towards-optimal-assessment-of-plausible","title":"B4: Towards Optimal Assessment of Plausible Code Solutions with Plausible Tests","date":"2024-09-13","arxiv_id":"2409.08692","repositories_listed":1,"syntology":null},{"url":"/paper/policy-filtration-in-rlhf-to-fine-tune-llm","slug":"policy-filtration-in-rlhf-to-fine-tune-llm","title":"Policy Filtration in RLHF to Fine-Tune LLM for Code Generation","date":"2024-09-11","arxiv_id":"2409.06957","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/policy-filtration-in-rlhf-to-fine-tune-llm#ran","syntology_url":"https://syntology.ai/paper/2409.06957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06957"}},"official":{"repos":["swtheing/pf-ppo-rlhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/diffqrcoder-diffusion-based-aesthetic-qr-code","slug":"diffqrcoder-diffusion-based-aesthetic-qr-code","title":"DiffQRCoder: Diffusion-based Aesthetic QR Code Generation with Scanning Robustness Guided Iterative Refinement","date":"2024-09-10","arxiv_id":"2409.06355","repositories_listed":1,"syntology":null},{"url":"/paper/hexacoder-secure-code-generation-via-oracle","slug":"hexacoder-secure-code-generation-via-oracle","title":"HexaCoder: Secure Code Generation via Oracle-Guided Synthetic Training Data","date":"2024-09-10","arxiv_id":"2409.06446","repositories_listed":1,"syntology":null},{"url":"/paper/a-pair-programming-framework-for-code","slug":"a-pair-programming-framework-for-code","title":"A Pair Programming Framework for Code Generation via Multi-Plan Exploration and Feedback-Driven Refinement","date":"2024-09-08","arxiv_id":"2409.05001","repositories_listed":1,"syntology":null},{"url":"/paper/insights-from-benchmarking-frontier-language","slug":"insights-from-benchmarking-frontier-language","title":"Insights from Benchmarking Frontier Language Models on Web App Code Generation","date":"2024-09-08","arxiv_id":"2409.05177","repositories_listed":1,"syntology":null},{"url":"/paper/multi-programming-language-ensemble-for-code","slug":"multi-programming-language-ensemble-for-code","title":"Multi-Programming Language Ensemble for Code Generation in Large Language Model","date":"2024-09-06","arxiv_id":"2409.04114","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-your-code-llms-perform-empowering-code","slug":"how-do-your-code-llms-perform-empowering-code","title":"How Do Your Code LLMs Perform? Empowering Code Instruction Tuning with High-Quality Data","date":"2024-09-05","arxiv_id":"2409.03810","repositories_listed":1,"syntology":null},{"url":"/paper/bioinformatics-retrieval-augmentation-data","slug":"bioinformatics-retrieval-augmentation-data","title":"Language Model Powered Digital Biology with BRAD","date":"2024-09-04","arxiv_id":"2409.02864","repositories_listed":1,"syntology":null},{"url":"/paper/robotwin-dual-arm-robot-benchmark-with","slug":"robotwin-dual-arm-robot-benchmark-with","title":"RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins (early version)","date":"2024-09-04","arxiv_id":"2409.02920","repositories_listed":1,"syntology":null},{"url":"/paper/it-is-time-to-develop-an-auditing-framework","slug":"it-is-time-to-develop-an-auditing-framework","title":"It is Time to Develop an Auditing Framework to Promote Value Aware Chatbots","date":"2024-09-03","arxiv_id":"2409.01539","repositories_listed":1,"syntology":null},{"url":"/paper/data-formulator-2-iteratively-creating-rich","slug":"data-formulator-2-iteratively-creating-rich","title":"Data Formulator 2: Iterative Creation of Data Visualizations, with AI Transforming Data Along the Way","date":"2024-08-28","arxiv_id":"2408.16119","repositories_listed":1,"syntology":null},{"url":"/paper/doce-finding-the-sweet-spot-for-execution","slug":"doce-finding-the-sweet-spot-for-execution","title":"DOCE: Finding the Sweet Spot for Execution-Based Code Generation","date":"2024-08-25","arxiv_id":"2408.13745","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doce-finding-the-sweet-spot-for-execution#ran","syntology_url":"https://syntology.ai/paper/2408.13745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13745"}},"official":{"repos":["deep-spin/doce"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-functional-correctness-enough-to-evaluate","slug":"is-functional-correctness-enough-to-evaluate","title":"Is Functional Correctness Enough to Evaluate Code Language Models? Exploring Diversity of Generated Codes","date":"2024-08-24","arxiv_id":"2408.14504","repositories_listed":1,"syntology":null},{"url":"/paper/codejudge-eval-can-large-language-models-be","slug":"codejudge-eval-can-large-language-models-be","title":"CodeJudge-Eval: Can Large Language Models be Good Judges in Code Understanding?","date":"2024-08-20","arxiv_id":"2408.10718","repositories_listed":1,"syntology":null},{"url":"/paper/epic-cost-effective-search-based-prompt","slug":"epic-cost-effective-search-based-prompt","title":"EPiC: Cost-effective Search-based Prompt Engineering of LLMs for Code Generation","date":"2024-08-20","arxiv_id":"2408.11198","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-verilogeval-newer-llms-in-context","slug":"revisiting-verilogeval-newer-llms-in-context","title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","date":"2024-08-20","arxiv_id":"2408.11053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-verilogeval-newer-llms-in-context#ran","syntology_url":"https://syntology.ai/paper/2408.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11053"}},"official":{"repos":["nvlabs/verilog-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selective-prompt-anchoring-for-code","slug":"selective-prompt-anchoring-for-code","title":"Selective Prompt Anchoring for Code Generation","date":"2024-08-17","arxiv_id":"2408.09121","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/selective-prompt-anchoring-for-code#ran","syntology_url":"https://syntology.ai/paper/2408.09121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09121"}},"official":{"repos":["magic-yuantian/selective-prompt-anchoring"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/i-sheep-self-alignment-of-llm-from-scratch","slug":"i-sheep-self-alignment-of-llm-from-scratch","title":"I-SHEEP: Self-Alignment of LLM from Scratch through an Iterative Self-Enhancement Paradigm","date":"2024-08-15","arxiv_id":"2408.08072","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/i-sheep-self-alignment-of-llm-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2408.08072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08072"}},"official":{"repos":["multimodal-art-projection/I-SHEEP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-the-code-debugging-ability-of-llms","slug":"enhancing-the-code-debugging-ability-of-llms","title":"COAST: Enhancing the Code Debugging Ability of LLMs through Communicative Agent Based Data Synthesis","date":"2024-08-09","arxiv_id":"2408.05006","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-rag-based-vulnerability","slug":"exploring-rag-based-vulnerability","title":"VulScribeR: Exploring RAG-based Vulnerability Augmentation with LLMs","date":"2024-08-07","arxiv_id":"2408.04125","repositories_listed":1,"syntology":null},{"url":"/paper/gui-element-detection-using-sota-yolo-deep","slug":"gui-element-detection-using-sota-yolo-deep","title":"GUI Element Detection Using SOTA YOLO Deep Learning Models","date":"2024-08-07","arxiv_id":"2408.03507","repositories_listed":1,"syntology":null},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01651","slug":"2408-01651","title":"Music2P: A Multi-Modal AI-Driven Tool for Simplifying Album Cover Design","date":"2024-08-03","arxiv_id":"2408.01651","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00989","slug":"2408-00989","title":"On the Resilience of LLM-Based Multi-Agent Collaboration with Faulty Agents","date":"2024-08-02","arxiv_id":"2408.00989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00989#ran","syntology_url":"https://syntology.ai/paper/2408.00989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00989"}},"official":{"repos":["cuhk-arise/mas-resilience"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00994","slug":"2408-00994","title":"ArchCode: Incorporating Software Requirements in Code Generation with Large Language Models","date":"2024-08-02","arxiv_id":"2408.00994","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01391","slug":"2408-01391","title":"FT K-means: A High-Performance K-means on GPU with Fault Tolerance","date":"2024-08-02","arxiv_id":"2408.01391","repositories_listed":1,"syntology":null},{"url":"/paper/autom3l-an-automated-multimodal-machine","slug":"autom3l-an-automated-multimodal-machine","title":"AutoM3L: An Automated Multimodal Machine Learning Framework with Large Language Models","date":"2024-08-01","arxiv_id":"2408.00665","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00019","slug":"2408-00019","title":"WebApp1K: A Practical Code-Generation Benchmark for Web App Development","date":"2024-07-30","arxiv_id":"2408.00019","repositories_listed":1,"syntology":null},{"url":"/paper/appworld-a-controllable-world-of-apps-and","slug":"appworld-a-controllable-world-of-apps-and","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","date":"2024-07-26","arxiv_id":"2407.18901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/appworld-a-controllable-world-of-apps-and#ran","syntology_url":"https://syntology.ai/paper/2407.18901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18901"}},"official":{"repos":["stonybrooknlp/appworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-pro-are-low-rank-adapters-properly","slug":"lora-pro-are-low-rank-adapters-properly","title":"LoRA-Pro: Are Low-Rank Adapters Properly Optimized?","date":"2024-07-25","arxiv_id":"2407.18242","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-pro-are-low-rank-adapters-properly#ran","syntology_url":"https://syntology.ai/paper/2407.18242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18242"}},"official":{"repos":["mrflogs/LoRA-Pro"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"597694000ed830a6093b758ca34f8aebc332505d23aa5d80ae236c9640f0f14f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}