{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/2","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":17,"rows_per_page":100,"rows":[101,200],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation","next":"/task/code-generation/papers/3","papers":[{"url":"/paper/doccoder-generating-code-by-retrieving-and","slug":"doccoder-generating-code-by-retrieving-and","title":"DocPrompting: Generating Code by Retrieving the Docs","date":"2022-07-13","arxiv_id":"2207.05987","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/doccoder-generating-code-by-retrieving-and#ran","syntology_url":"https://syntology.ai/paper/2207.05987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05987"}},"official":{"repos":["shuyanzhou/doccoder","shuyanzhou/docprompting"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/natgen-generative-pre-training-by","slug":"natgen-generative-pre-training-by","title":"NatGen: Generative pre-training by \"Naturalizing\" source code","date":"2022-06-15","arxiv_id":"2206.07585","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natgen-generative-pre-training-by#ran","syntology_url":"https://syntology.ai/paper/2206.07585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07585"}},"official":{"repos":["saikat107/natgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codeattack-code-based-adversarial-attacks-for","slug":"codeattack-code-based-adversarial-attacks-for","title":"CodeAttack: Code-Based Adversarial Attacks for Pre-trained Programming Language Models","date":"2022-05-31","arxiv_id":"2206.00052","repositories_listed":2,"syntology":null},{"url":"/paper/the-impact-of-lexical-and-grammatical-1","slug":"the-impact-of-lexical-and-grammatical-1","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2022-02-28","arxiv_id":"2202.13972","repositories_listed":2,"syntology":null},{"url":"/paper/competition-level-code-generation-with-1","slug":"competition-level-code-generation-with-1","title":"Competition-Level Code Generation with AlphaCode","date":"2022-02-08","arxiv_id":"2203.07814","repositories_listed":2,"syntology":null},{"url":"/paper/synchromesh-reliable-code-generation-from-pre-1","slug":"synchromesh-reliable-code-generation-from-pre-1","title":"Synchromesh: Reliable code generation from pre-trained language models","date":"2022-01-26","arxiv_id":"2201.11227","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synchromesh-reliable-code-generation-from-pre-1#ran","syntology_url":"https://syntology.ai/paper/2201.11227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11227"}},"official":null}},{"url":"/paper/lamda-language-models-for-dialog-applications","slug":"lamda-language-models-for-dialog-applications","title":"LaMDA: Language Models for Dialog Applications","date":"2022-01-20","arxiv_id":"2201.08239","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lamda-language-models-for-dialog-applications#ran","syntology_url":"https://syntology.ai/paper/2201.08239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08239"}},"official":null}},{"url":"/paper/controlling-conditional-language-models-with","slug":"controlling-conditional-language-models-with","title":"Controlling Conditional Language Models without Catastrophic Forgetting","date":"2021-12-01","arxiv_id":"2112.00791","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":1,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/controlling-conditional-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2112.00791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.00791"}},"official":{"repos":["naver/gdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/salience-guided-iterative-asymmetric-mutual","slug":"salience-guided-iterative-asymmetric-mutual","title":"Salience-Guided Iterative Asymmetric Mutual Hashing for Fast Person Re-identification","date":"2021-09-08","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/retrieval-augmented-code-generation-and","slug":"retrieval-augmented-code-generation-and","title":"Retrieval Augmented Code Generation and Summarization","date":"2021-08-26","arxiv_id":"2108.11601","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieval-augmented-code-generation-and#ran","syntology_url":"https://syntology.ai/paper/2108.11601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.11601"}},"official":{"repos":["rizwan09/redcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/energy-based-models-for-code-generation-under","slug":"energy-based-models-for-code-generation-under","title":"Energy-Based Models for Code Generation under Compilability Constraints","date":"2021-06-09","arxiv_id":"2106.04985","repositories_listed":2,"syntology":null},{"url":"/paper/text2app-a-framework-for-creating-android","slug":"text2app-a-framework-for-creating-android","title":"Text2App: A Framework for Creating Android Apps from Text Descriptions","date":"2021-04-16","arxiv_id":"2104.08301","repositories_listed":2,"syntology":null},{"url":"/paper/unified-pre-training-for-program","slug":"unified-pre-training-for-program","title":"Unified Pre-training for Program Understanding and Generation","date":"2021-03-10","arxiv_id":"2103.06333","repositories_listed":2,"syntology":null},{"url":"/paper/object-detection-for-graphical-user-interface","slug":"object-detection-for-graphical-user-interface","title":"Object Detection for Graphical User Interface: Old Fashioned or Deep Learning or a Combination?","date":"2020-08-12","arxiv_id":"2008.05132","repositories_listed":2,"syntology":null},{"url":"/paper/graph-based-self-supervised-program-repair","slug":"graph-based-self-supervised-program-repair","title":"Graph-based, Self-Supervised Program Repair from Diagnostic Feedback","date":"2020-05-20","arxiv_id":"2005.10636","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/graph-based-self-supervised-program-repair#ran","syntology_url":"https://syntology.ai/paper/2005.10636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.10636"}},"official":{"repos":["michiyasunaga/DrRepair","worksheets.codalab.org/worksheets/0x01838644724a433c932bef4cb5c42fbd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/incorporating-external-knowledge-through-pre","slug":"incorporating-external-knowledge-through-pre","title":"Incorporating External Knowledge through Pre-training for Natural Language to Code Generation","date":"2020-04-20","arxiv_id":"2004.09015","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/incorporating-external-knowledge-through-pre#ran","syntology_url":"https://syntology.ai/paper/2004.09015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09015"}},"official":{"repos":["neulab/external-knowledge-codegen"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/treegen-a-tree-based-transformer-architecture","slug":"treegen-a-tree-based-transformer-architecture","title":"TreeGen: A Tree-Based Transformer Architecture for Code Generation","date":"2019-11-22","arxiv_id":"1911.09983","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treegen-a-tree-based-transformer-architecture#ran","syntology_url":"https://syntology.ai/paper/1911.09983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.09983"}},"official":{"repos":["zysszy/TreeGen"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/code-generation-as-a-dual-task-of-code","slug":"code-generation-as-a-dual-task-of-code","title":"Code Generation as a Dual Task of Code Summarization","date":"2019-10-14","arxiv_id":"1910.05923","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-generation-as-a-dual-task-of-code#ran","syntology_url":"https://syntology.ai/paper/1910.05923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05923"}},"official":{"repos":["Bolin0215/CSCGDual"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/juice-a-large-scale-distantly-supervised","slug":"juice-a-large-scale-distantly-supervised","title":"JuICe: A Large Scale Distantly Supervised Dataset for Open Domain Context-based Code Generation","date":"2019-10-05","arxiv_id":"1910.02216","repositories_listed":2,"syntology":null},{"url":"/paper/structural-language-models-for-any-code","slug":"structural-language-models-for-any-code","title":"Structural Language Models of Code","date":"2019-09-30","arxiv_id":"1910.00577","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/structural-language-models-for-any-code#ran","syntology_url":"https://syntology.ai/paper/1910.00577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00577"}},"official":{"repos":["tech-srl/slm-code-generation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/building-language-models-for-text-with-named","slug":"building-language-models-for-text-with-named","title":"Building Language Models for Text with Named Entities","date":"2018-05-13","arxiv_id":"1805.04836","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/building-language-models-for-text-with-named#ran","syntology_url":"https://syntology.ai/paper/1805.04836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.04836"}},"official":{"repos":["uclanlp/NamedEntityLanguageModel"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/bidirectional-attention-for-sql-generation","slug":"bidirectional-attention-for-sql-generation","title":"Bidirectional Attention for SQL Generation","date":"2017-12-30","arxiv_id":"1801.00076","repositories_listed":2,"syntology":null},{"url":"/paper/latent-predictor-networks-for-code-generation","slug":"latent-predictor-networks-for-code-generation","title":"Latent Predictor Networks for Code Generation","date":"2016-03-22","arxiv_id":"1603.06744","repositories_listed":2,"syntology":null},{"url":"/paper/the-devil-behind-the-mask-an-emergent-safety","slug":"the-devil-behind-the-mask-an-emergent-safety","title":"The Devil behind the mask: An emergent safety vulnerability of Diffusion LLMs","date":"2025-07-15","arxiv_id":"2507.11097","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-devil-behind-the-mask-an-emergent-safety#ran","syntology_url":"https://syntology.ai/paper/2507.11097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11097"}},"official":{"repos":["zichenwen1/dija"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kodezi-chronos-a-debugging-first-language","slug":"kodezi-chronos-a-debugging-first-language","title":"Kodezi Chronos: A Debugging-First Language Model for Repository-Scale, Memory-Driven Code Understanding","date":"2025-07-14","arxiv_id":"2507.12482","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-verification-for-llm-code","slug":"rethinking-verification-for-llm-code","title":"Rethinking Verification for LLM Code Generation: From Generation to Testing","date":"2025-07-09","arxiv_id":"2507.06920","repositories_listed":1,"syntology":null},{"url":"/paper/corecodebench-a-configurable-multi-scenario","slug":"corecodebench-a-configurable-multi-scenario","title":"CoreCodeBench: A Configurable Multi-Scenario Repository-Level Benchmark","date":"2025-07-04","arxiv_id":"2507.05281","repositories_listed":1,"syntology":null},{"url":"/paper/evoagentx-an-automated-framework-for-evolving","slug":"evoagentx-an-automated-framework-for-evolving","title":"EvoAgentX: An Automated Framework for Evolving Agentic Workflows","date":"2025-07-04","arxiv_id":"2507.03616","repositories_listed":1,"syntology":{"n":20,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":16,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evoagentx-an-automated-framework-for-evolving#ran","syntology_url":"https://syntology.ai/paper/2507.03616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03616"}},"official":{"repos":["evoagentx/evoagentx"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/estimating-correctness-without-oracles-in-llm","slug":"estimating-correctness-without-oracles-in-llm","title":"Estimating Correctness Without Oracles in LLM-Based Code Generation","date":"2025-06-26","arxiv_id":"2507.00057","repositories_listed":1,"syntology":null},{"url":"/paper/diffucoder-understanding-and-improving-masked","slug":"diffucoder-understanding-and-improving-masked","title":"DiffuCoder: Understanding and Improving Masked Diffusion Models for Code Generation","date":"2025-06-25","arxiv_id":"2506.20639","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffucoder-understanding-and-improving-masked#ran","syntology_url":"https://syntology.ai/paper/2506.20639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20639"}},"official":{"repos":["apple/ml-diffucoder"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-modeling-by-language-models","slug":"language-modeling-by-language-models","title":"Language Modeling by Language Models","date":"2025-06-25","arxiv_id":"2506.20249","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-modeling-by-language-models#ran","syntology_url":"https://syntology.ai/paper/2506.20249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20249"}},"official":{"repos":["allenai/genesys"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/recode-updating-code-api-knowledge-with","slug":"recode-updating-code-api-knowledge-with","title":"ReCode: Updating Code API Knowledge with Reinforcement Learning","date":"2025-06-25","arxiv_id":"2506.20495","repositories_listed":1,"syntology":null},{"url":"/paper/from-reproduction-to-replication-evaluating","slug":"from-reproduction-to-replication-evaluating","title":"From Reproduction to Replication: Evaluating Research Agents with Progressive Code Masking","date":"2025-06-24","arxiv_id":"2506.19724","repositories_listed":1,"syntology":null},{"url":"/paper/texpert-a-multi-level-benchmark-for","slug":"texpert-a-multi-level-benchmark-for","title":"TeXpert: A Multi-Level Benchmark for Evaluating LaTeX Code Generation by LLMs","date":"2025-06-20","arxiv_id":"2506.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/texpert-a-multi-level-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.16990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.16990"}},"official":{"repos":["knowledge-verse-ai/texpert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cast-enhancing-code-retrieval-augmented","slug":"cast-enhancing-code-retrieval-augmented","title":"cAST: Enhancing Code Retrieval-Augmented Generation with Structural Chunking via Abstract Syntax Tree","date":"2025-06-18","arxiv_id":"2506.15655","repositories_listed":1,"syntology":null},{"url":"/paper/comprehensive-verilog-design-problems-a-next","slug":"comprehensive-verilog-design-problems-a-next","title":"Comprehensive Verilog Design Problems: A Next-Generation Benchmark Dataset for Evaluating Large Language Models and Agents on RTL Design and Verification","date":"2025-06-17","arxiv_id":"2506.14074","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-from-your-language-model-one-byte-at","slug":"sampling-from-your-language-model-one-byte-at","title":"Sampling from Your Language Model One Byte at a Time","date":"2025-06-17","arxiv_id":"2506.14123","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/sampling-from-your-language-model-one-byte-at#ran","syntology_url":"https://syntology.ai/paper/2506.14123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14123"}},"official":{"repos":["sewoonglab/byte-sampler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xolver-multi-agent-reasoning-with-holistic","slug":"xolver-multi-agent-reasoning-with-holistic","title":"Xolver: Multi-Agent Reasoning with Holistic Experience Learning Just Like an Olympiad Team","date":"2025-06-17","arxiv_id":"2506.14234","repositories_listed":1,"syntology":null},{"url":"/paper/locationreasoner-evaluating-llms-on-real","slug":"locationreasoner-evaluating-llms-on-real","title":"LocationReasoner: Evaluating LLMs on Real-World Site Selection Reasoning","date":"2025-06-16","arxiv_id":"2506.13841","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locationreasoner-evaluating-llms-on-real#ran","syntology_url":"https://syntology.ai/paper/2506.13841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13841"}},"official":{"repos":["miho-koda/locationreasoner"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/humanity-s-last-code-exam-can-advanced-llms","slug":"humanity-s-last-code-exam-can-advanced-llms","title":"Humanity's Last Code Exam: Can Advanced LLMs Conquer Human's Hardest Code Competition?","date":"2025-06-15","arxiv_id":"2506.12713","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanity-s-last-code-exam-can-advanced-llms#ran","syntology_url":"https://syntology.ai/paper/2506.12713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12713"}},"official":{"repos":["humanity-s-last-code-exam/hlce"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/pro-v-an-efficient-program-generation-multi","slug":"pro-v-an-efficient-program-generation-multi","title":"PRO-V: An Efficient Program Generation Multi-Agent System for Automatic RTL Verification","date":"2025-06-13","arxiv_id":"2506.12200","repositories_listed":1,"syntology":null},{"url":"/paper/swe-bench-cl-continual-learning-for-coding","slug":"swe-bench-cl-continual-learning-for-coding","title":"SWE-Bench-CL: Continual Learning for Coding Agents","date":"2025-06-13","arxiv_id":"2507.00014","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/swe-bench-cl-continual-learning-for-coding#ran","syntology_url":"https://syntology.ai/paper/2507.00014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.00014"}},"official":{"repos":["thomasjoshi/agents-never-forget"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-10467","slug":"2506-10467","title":"Specification and Evaluation of Multi-Agent LLM Systems -- Prototype and Cybersecurity Applications","date":"2025-06-12","arxiv_id":"2506.10467","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10974","slug":"2506-10974","title":"AutoMind: Adaptive Knowledgeable Agent for Automated Data Science","date":"2025-06-12","arxiv_id":"2506.10974","repositories_listed":1,"syntology":null},{"url":"/paper/execution-guided-line-by-line-code-generation","slug":"execution-guided-line-by-line-code-generation","title":"Execution Guided Line-by-Line Code Generation","date":"2025-06-12","arxiv_id":"2506.10948","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-open-source-and","slug":"bridging-the-gap-between-open-source-and","title":"Bridging the Gap Between Open-Source and Proprietary LLMs in Table QA","date":"2025-06-11","arxiv_id":"2506.09657","repositories_listed":1,"syntology":null},{"url":"/paper/vicrit-a-verifiable-reinforcement-learning","slug":"vicrit-a-verifiable-reinforcement-learning","title":"ViCrit: A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs","date":"2025-06-11","arxiv_id":"2506.10128","repositories_listed":1,"syntology":null},{"url":"/paper/utboost-rigorous-evaluation-of-coding-agents","slug":"utboost-rigorous-evaluation-of-coding-agents","title":"UTBoost: Rigorous Evaluation of Coding Agents on SWE-Bench","date":"2025-06-10","arxiv_id":"2506.09289","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10022","slug":"2506-10022","title":"LLMs Caught in the Crossfire: Malware Requests and Jailbreak Challenges","date":"2025-06-09","arxiv_id":"2506.10022","repositories_listed":1,"syntology":null},{"url":"/paper/slidecoder-layout-aware-rag-enhanced","slug":"slidecoder-layout-aware-rag-enhanced","title":"SlideCoder: Layout-aware RAG-enhanced Hierarchical Slide Generation from Design","date":"2025-06-09","arxiv_id":"2506.07964","repositories_listed":1,"syntology":null},{"url":"/paper/2506-06971","slug":"2506-06971","title":"Chain-of-Code Collapse: Reasoning Failures in LLMs via Adversarial Prompting in Code Generation","date":"2025-06-08","arxiv_id":"2506.06971","repositories_listed":1,"syntology":null},{"url":"/paper/designbench-a-comprehensive-benchmark-for","slug":"designbench-a-comprehensive-benchmark-for","title":"DesignBench: A Comprehensive Benchmark for MLLM-based Front-end Code Generation","date":"2025-06-06","arxiv_id":"2506.06251","repositories_listed":1,"syntology":null},{"url":"/paper/kramabench-a-benchmark-for-ai-systems-on-data","slug":"kramabench-a-benchmark-for-ai-systems-on-data","title":"KramaBench: A Benchmark for AI Systems on Data-to-Insight Pipelines over Data Lakes","date":"2025-06-06","arxiv_id":"2506.06541","repositories_listed":1,"syntology":null},{"url":"/paper/deployability-centric-infrastructure-as-code","slug":"deployability-centric-infrastructure-as-code","title":"Deployability-Centric Infrastructure-as-Code Generation: An LLM-based Iterative Framework","date":"2025-06-05","arxiv_id":"2506.05623","repositories_listed":1,"syntology":null},{"url":"/paper/icpc-eval-probing-the-frontiers-of-llm","slug":"icpc-eval-probing-the-frontiers-of-llm","title":"ICPC-Eval: Probing the Frontiers of LLM Reasoning with Competitive Programming Contests","date":"2025-06-05","arxiv_id":"2506.04894","repositories_listed":1,"syntology":null},{"url":"/paper/matter-of-fact-a-benchmark-for-verifying-the","slug":"matter-of-fact-a-benchmark-for-verifying-the","title":"Matter-of-Fact: A Benchmark for Verifying the Feasibility of Literature-Supported Claims in Materials Science","date":"2025-06-04","arxiv_id":"2506.04410","repositories_listed":1,"syntology":null},{"url":"/paper/seed-coder-let-the-code-model-curate-data-for","slug":"seed-coder-let-the-code-model-curate-data-for","title":"Seed-Coder: Let the Code Model Curate Data for Itself","date":"2025-06-04","arxiv_id":"2506.03524","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-graph-pruning-for-multi-agent","slug":"adaptive-graph-pruning-for-multi-agent","title":"Adaptive Graph Pruning for Multi-Agent Communication","date":"2025-06-03","arxiv_id":"2506.02951","repositories_listed":1,"syntology":null},{"url":"/paper/diablo-diagonal-blocks-are-sufficient-for","slug":"diablo-diagonal-blocks-are-sufficient-for","title":"DiaBlo: Diagonal Blocks Are Sufficient For Finetuning","date":"2025-06-03","arxiv_id":"2506.03230","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diablo-diagonal-blocks-are-sufficient-for#ran","syntology_url":"https://syntology.ai/paper/2506.03230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03230"}},"official":{"repos":["ziyangjoy/diablo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rmoa-optimizing-mixture-of-agents-through","slug":"rmoa-optimizing-mixture-of-agents-through","title":"RMoA: Optimizing Mixture-of-Agents through Diversity Maximization and Residual Compensation","date":"2025-05-30","arxiv_id":"2505.24442","repositories_listed":1,"syntology":null},{"url":"/paper/llm-performance-for-code-generation-on-noisy","slug":"llm-performance-for-code-generation-on-noisy","title":"LLM Performance for Code Generation on Noisy Tasks","date":"2025-05-29","arxiv_id":"2505.23598","repositories_listed":1,"syntology":null},{"url":"/paper/oss-uagent-an-agent-based-usability","slug":"oss-uagent-an-agent-based-usability","title":"OSS-UAgent: An Agent-based Usability Evaluation Framework for Open Source Software","date":"2025-05-29","arxiv_id":"2505.23239","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-code-generation-using-small","slug":"self-correcting-code-generation-using-small","title":"Self-Correcting Code Generation Using Small Language Models","date":"2025-05-29","arxiv_id":"2505.23060","repositories_listed":1,"syntology":null},{"url":"/paper/verina-benchmarking-verifiable-code","slug":"verina-benchmarking-verifiable-code","title":"VERINA: Benchmarking Verifiable Code Generation","date":"2025-05-29","arxiv_id":"2505.23135","repositories_listed":1,"syntology":null},{"url":"/paper/training-language-models-to-generate-quality","slug":"training-language-models-to-generate-quality","title":"Training Language Models to Generate Quality Code with Program Analysis Feedback","date":"2025-05-28","arxiv_id":"2505.22704","repositories_listed":1,"syntology":null},{"url":"/paper/r1-code-interpreter-training-llms-to-reason","slug":"r1-code-interpreter-training-llms-to-reason","title":"R1-Code-Interpreter: Training LLMs to Reason with Code via Supervised and Reinforcement Learning","date":"2025-05-27","arxiv_id":"2505.21668","repositories_listed":1,"syntology":null},{"url":"/paper/repomaster-autonomous-exploration-and","slug":"repomaster-autonomous-exploration-and","title":"RepoMaster: Autonomous Exploration and Understanding of GitHub Repositories for Complex Task Solving","date":"2025-05-27","arxiv_id":"2505.21577","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-strong-weak-model","slug":"an-empirical-study-on-strong-weak-model","title":"An Empirical Study on Strong-Weak Model Collaboration for Repo-level Code Generation","date":"2025-05-26","arxiv_id":"2505.20182","repositories_listed":1,"syntology":null},{"url":"/paper/compliance-to-code-enhancing-financial","slug":"compliance-to-code-enhancing-financial","title":"Compliance-to-Code: Enhancing Financial Compliance Checking via Code Generation","date":"2025-05-26","arxiv_id":"2505.19804","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-without-external-rewards","slug":"learning-to-reason-without-external-rewards","title":"Learning to Reason without External Rewards","date":"2025-05-26","arxiv_id":"2505.19590","repositories_listed":1,"syntology":null},{"url":"/paper/rechisel-effective-automatic-chisel-code","slug":"rechisel-effective-automatic-chisel-code","title":"ReChisel: Effective Automatic Chisel Code Generation by LLM with Reflection","date":"2025-05-26","arxiv_id":"2505.19734","repositories_listed":1,"syntology":null},{"url":"/paper/style2code-a-style-controllable-code","slug":"style2code-a-style-controllable-code","title":"Style2Code: A Style-Controllable Code Generation Framework with Dual-Modal Contrastive Representation Learning","date":"2025-05-26","arxiv_id":"2505.19442","repositories_listed":1,"syntology":null},{"url":"/paper/swe-rebench-an-automated-pipeline-for-task","slug":"swe-rebench-an-automated-pipeline-for-task","title":"SWE-rebench: An Automated Pipeline for Task Collection and Decontaminated Evaluation of Software Engineering Agents","date":"2025-05-26","arxiv_id":"2505.20411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swe-rebench-an-automated-pipeline-for-task#ran","syntology_url":"https://syntology.ai/paper/2505.20411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20411"}},"official":{"repos":["swe-rebench/swe-bench-fork"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/win-fast-or-lose-slow-balancing-speed-and","slug":"win-fast-or-lose-slow-balancing-speed-and","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","date":"2025-05-26","arxiv_id":"2505.19481","repositories_listed":1,"syntology":null},{"url":"/paper/mind-the-gap-a-practical-attack-on-gguf","slug":"mind-the-gap-a-practical-attack-on-gguf","title":"Mind the Gap: A Practical Attack on GGUF Quantization","date":"2025-05-24","arxiv_id":"2505.23786","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mind-the-gap-a-practical-attack-on-gguf#ran","syntology_url":"https://syntology.ai/paper/2505.23786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23786"}},"official":{"repos":["eth-sri/llm-quantization-attack"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/sew-self-evolving-agentic-workflows-for","slug":"sew-self-evolving-agentic-workflows-for","title":"SEW: Self-Evolving Agentic Workflows for Automated Code Generation","date":"2025-05-24","arxiv_id":"2505.18646","repositories_listed":1,"syntology":null},{"url":"/paper/fullfront-benchmarking-mllms-across-the-full","slug":"fullfront-benchmarking-mllms-across-the-full","title":"FullFront: Benchmarking MLLMs Across the Full Front-End Engineering Workflow","date":"2025-05-23","arxiv_id":"2505.17399","repositories_listed":1,"syntology":null},{"url":"/paper/hygenar-an-llm-driven-hybrid-genetic","slug":"hygenar-an-llm-driven-hybrid-genetic","title":"HyGenar: An LLM-Driven Hybrid Genetic Algorithm for Few-Shot Grammar Generation","date":"2025-05-22","arxiv_id":"2505.16978","repositories_listed":1,"syntology":null},{"url":"/paper/cad-coder-an-open-source-vision-language","slug":"cad-coder-an-open-source-vision-language","title":"CAD-Coder: An Open-Source Vision-Language Model for Computer-Aided Design Code Generation","date":"2025-05-20","arxiv_id":"2505.14646","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cad-coder-an-open-source-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.14646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14646"}},"official":{"repos":["anniedoris/cad-coder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mlzero-a-multi-agent-system-for-end-to-end","slug":"mlzero-a-multi-agent-system-for-end-to-end","title":"MLZero: A Multi-Agent System for End-to-end Machine Learning Automation","date":"2025-05-20","arxiv_id":"2505.13941","repositories_listed":1,"syntology":null},{"url":"/paper/ad-agent-a-multi-agent-framework-for-end-to","slug":"ad-agent-a-multi-agent-framework-for-end-to","title":"AD-AGENT: A Multi-agent Framework for End-to-end Anomaly Detection","date":"2025-05-19","arxiv_id":"2505.12594","repositories_listed":1,"syntology":null},{"url":"/paper/agi-elo-how-far-are-we-from-mastering-a-task","slug":"agi-elo-how-far-are-we-from-mastering-a-task","title":"AGI-Elo: How Far Are We From Mastering A Task?","date":"2025-05-19","arxiv_id":"2505.12844","repositories_listed":1,"syntology":null},{"url":"/paper/effibench-x-a-multi-language-benchmark-for","slug":"effibench-x-a-multi-language-benchmark-for","title":"EffiBench-X: A Multi-Language Benchmark for Measuring Efficiency of LLM-Generated Code","date":"2025-05-19","arxiv_id":"2505.13004","repositories_listed":1,"syntology":null},{"url":"/paper/rn-f-a-novel-approach-for-mitigating","slug":"rn-f-a-novel-approach-for-mitigating","title":"RN-F: A Novel Approach for Mitigating Contaminated Data in Large Language Models","date":"2025-05-19","arxiv_id":"2505.13249","repositories_listed":1,"syntology":null},{"url":"/paper/halo-hierarchical-autonomous-logic-oriented","slug":"halo-hierarchical-autonomous-logic-oriented","title":"HALO: Hierarchical Autonomous Logic-Oriented Orchestration for Multi-Agent LLM Systems","date":"2025-05-17","arxiv_id":"2505.13516","repositories_listed":1,"syntology":null},{"url":"/paper/omac-a-broad-optimization-framework-for-llm","slug":"omac-a-broad-optimization-framework-for-llm","title":"OMAC: A Broad Optimization Framework for LLM-Based Multi-Agent Collaboration","date":"2025-05-17","arxiv_id":"2505.11765","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/omac-a-broad-optimization-framework-for-llm#ran","syntology_url":"https://syntology.ai/paper/2505.11765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11765"}},"official":{"repos":["xiwenchao/omac-demo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/verireason-reinforcement-learning-with-1","slug":"verireason-reinforcement-learning-with-1","title":"VeriReason: Reinforcement Learning with Testbench Feedback for Reasoning-Enhanced Verilog Generation","date":"2025-05-17","arxiv_id":"2505.11849","repositories_listed":1,"syntology":null},{"url":"/paper/verithoughts-enabling-automated-verilog-code","slug":"verithoughts-enabling-automated-verilog-code","title":"VeriThoughts: Enabling Automated Verilog Code Generation using Reasoning and Formal Verification","date":"2025-05-16","arxiv_id":"2505.20302","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verithoughts-enabling-automated-verilog-code#ran","syntology_url":"https://syntology.ai/paper/2505.20302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20302"}},"official":{"repos":["wilyub/verithoughts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10607","slug":"2505-10607","title":"MONAQ: Multi-Objective Neural Architecture Querying for Time-Series Analysis on Resource-Constrained Devices","date":"2025-05-15","arxiv_id":"2505.10607","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2505-10607#ran","syntology_url":"https://syntology.ai/paper/2505.10607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10607"}},"official":{"repos":["kaist-dmlab/monaq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/are-sparse-autoencoders-useful-for-java","slug":"are-sparse-autoencoders-useful-for-java","title":"Are Sparse Autoencoders Useful for Java Function Bug Detection?","date":"2025-05-15","arxiv_id":"2505.10375","repositories_listed":1,"syntology":null},{"url":"/paper/can-you-really-trust-code-copilots-evaluating","slug":"can-you-really-trust-code-copilots-evaluating","title":"Can You Really Trust Code Copilots? Evaluating Large Language Models from a Code Security Perspective","date":"2025-05-15","arxiv_id":"2505.10494","repositories_listed":1,"syntology":null},{"url":"/paper/complexformer-disruptively-advancing","slug":"complexformer-disruptively-advancing","title":"ComplexFormer: Disruptively Advancing Transformer Inference Ability via Head-Specific Complex Vector Attention","date":"2025-05-15","arxiv_id":"2505.10222","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-repetition-problems-of-llms-in","slug":"rethinking-repetition-problems-of-llms-in","title":"Rethinking Repetition Problems of LLMs in Code Generation","date":"2025-05-15","arxiv_id":"2505.10402","repositories_listed":1,"syntology":null},{"url":"/paper/codepde-an-inference-framework-for-llm-driven","slug":"codepde-an-inference-framework-for-llm-driven","title":"CodePDE: An Inference Framework for LLM-driven PDE Solver Generation","date":"2025-05-13","arxiv_id":"2505.08783","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-code-generation-via-bidirectional","slug":"enhancing-code-generation-via-bidirectional","title":"Enhancing Code Generation via Bidirectional Comment-Level Mutual Grounding","date":"2025-05-12","arxiv_id":"2505.07768","repositories_listed":1,"syntology":null},{"url":"/paper/web-bench-a-llm-code-benchmark-based-on-web","slug":"web-bench-a-llm-code-benchmark-based-on-web","title":"Web-Bench: A LLM Code Benchmark Based on Web Standards and Frameworks","date":"2025-05-12","arxiv_id":"2505.07473","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ensuring-reproducibility-in-generative-ai","slug":"ensuring-reproducibility-in-generative-ai","title":"Ensuring Reproducibility in Generative AI Systems for General Use Cases: A Framework for Regression Testing and Open Datasets","date":"2025-05-02","arxiv_id":"2505.02854","repositories_listed":1,"syntology":null},{"url":"/paper/program-semantic-inequivalence-game-with","slug":"program-semantic-inequivalence-game-with","title":"Program Semantic Inequivalence Game with Large Language Models","date":"2025-05-02","arxiv_id":"2505.03818","repositories_listed":1,"syntology":null}],"record_sha256":"dd2f3e578a1a52feb98a2dd51d8042b4b25478f1255afe5e252fba7b15217782","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}