{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/ran/1","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":280,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation/papers/ran/1","prev":null,"next":"/task/code-generation/papers/ran/2","papers":[{"url":"/paper/the-devil-behind-the-mask-an-emergent-safety","slug":"the-devil-behind-the-mask-an-emergent-safety","title":"The Devil behind the mask: An emergent safety vulnerability of Diffusion LLMs","date":"2025-07-15","arxiv_id":"2507.11097","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-devil-behind-the-mask-an-emergent-safety#ran","syntology_url":"https://syntology.ai/paper/2507.11097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11097"}},"official":{"repos":["zichenwen1/dija"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codeassistbench-cab-dataset-benchmarking-for","slug":"codeassistbench-cab-dataset-benchmarking-for","title":"CodeAssistBench (CAB): Dataset & Benchmarking for Multi-turn Chat-Based Code Assistance","date":"2025-07-14","arxiv_id":"2507.10646","repositories_listed":0,"syntology":{"n":24,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/codeassistbench-cab-dataset-benchmarking-for#ran","syntology_url":"https://syntology.ai/paper/2507.10646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10646"}},"official":null}},{"url":"/paper/evoagentx-an-automated-framework-for-evolving","slug":"evoagentx-an-automated-framework-for-evolving","title":"EvoAgentX: An Automated Framework for Evolving Agentic Workflows","date":"2025-07-04","arxiv_id":"2507.03616","repositories_listed":1,"syntology":{"n":20,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":16,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evoagentx-an-automated-framework-for-evolving#ran","syntology_url":"https://syntology.ai/paper/2507.03616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03616"}},"official":{"repos":["evoagentx/evoagentx"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/language-modeling-by-language-models","slug":"language-modeling-by-language-models","title":"Language Modeling by Language Models","date":"2025-06-25","arxiv_id":"2506.20249","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-modeling-by-language-models#ran","syntology_url":"https://syntology.ai/paper/2506.20249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20249"}},"official":{"repos":["allenai/genesys"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffucoder-understanding-and-improving-masked","slug":"diffucoder-understanding-and-improving-masked","title":"DiffuCoder: Understanding and Improving Masked Diffusion Models for Code Generation","date":"2025-06-25","arxiv_id":"2506.20639","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffucoder-understanding-and-improving-masked#ran","syntology_url":"https://syntology.ai/paper/2506.20639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20639"}},"official":{"repos":["apple/ml-diffucoder"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/texpert-a-multi-level-benchmark-for","slug":"texpert-a-multi-level-benchmark-for","title":"TeXpert: A Multi-Level Benchmark for Evaluating LaTeX Code Generation by LLMs","date":"2025-06-20","arxiv_id":"2506.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/texpert-a-multi-level-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.16990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.16990"}},"official":{"repos":["knowledge-verse-ai/texpert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sampling-from-your-language-model-one-byte-at","slug":"sampling-from-your-language-model-one-byte-at","title":"Sampling from Your Language Model One Byte at a Time","date":"2025-06-17","arxiv_id":"2506.14123","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/sampling-from-your-language-model-one-byte-at#ran","syntology_url":"https://syntology.ai/paper/2506.14123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14123"}},"official":{"repos":["sewoonglab/byte-sampler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/locationreasoner-evaluating-llms-on-real","slug":"locationreasoner-evaluating-llms-on-real","title":"LocationReasoner: Evaluating LLMs on Real-World Site Selection Reasoning","date":"2025-06-16","arxiv_id":"2506.13841","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locationreasoner-evaluating-llms-on-real#ran","syntology_url":"https://syntology.ai/paper/2506.13841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13841"}},"official":{"repos":["miho-koda/locationreasoner"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/humanity-s-last-code-exam-can-advanced-llms","slug":"humanity-s-last-code-exam-can-advanced-llms","title":"Humanity's Last Code Exam: Can Advanced LLMs Conquer Human's Hardest Code Competition?","date":"2025-06-15","arxiv_id":"2506.12713","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanity-s-last-code-exam-can-advanced-llms#ran","syntology_url":"https://syntology.ai/paper/2506.12713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12713"}},"official":{"repos":["humanity-s-last-code-exam/hlce"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/swe-bench-cl-continual-learning-for-coding","slug":"swe-bench-cl-continual-learning-for-coding","title":"SWE-Bench-CL: Continual Learning for Coding Agents","date":"2025-06-13","arxiv_id":"2507.00014","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/swe-bench-cl-continual-learning-for-coding#ran","syntology_url":"https://syntology.ai/paper/2507.00014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.00014"}},"official":{"repos":["thomasjoshi/agents-never-forget"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-evolving-llm-coder-and-unit-tester-via","slug":"co-evolving-llm-coder-and-unit-tester-via","title":"Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning","date":"2025-06-03","arxiv_id":"2506.03136","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-evolving-llm-coder-and-unit-tester-via#ran","syntology_url":"https://syntology.ai/paper/2506.03136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03136"}},"official":{"repos":["gen-verse/cure"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/diablo-diagonal-blocks-are-sufficient-for","slug":"diablo-diagonal-blocks-are-sufficient-for","title":"DiaBlo: Diagonal Blocks Are Sufficient For Finetuning","date":"2025-06-03","arxiv_id":"2506.03230","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diablo-diagonal-blocks-are-sufficient-for#ran","syntology_url":"https://syntology.ai/paper/2506.03230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03230"}},"official":{"repos":["ziyangjoy/diablo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/swe-rebench-an-automated-pipeline-for-task","slug":"swe-rebench-an-automated-pipeline-for-task","title":"SWE-rebench: An Automated Pipeline for Task Collection and Decontaminated Evaluation of Software Engineering Agents","date":"2025-05-26","arxiv_id":"2505.20411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swe-rebench-an-automated-pipeline-for-task#ran","syntology_url":"https://syntology.ai/paper/2505.20411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20411"}},"official":{"repos":["swe-rebench/swe-bench-fork"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chartgalaxy-a-dataset-for-infographic-chart","slug":"chartgalaxy-a-dataset-for-infographic-chart","title":"ChartGalaxy: A Dataset for Infographic Chart Understanding and Generation","date":"2025-05-24","arxiv_id":"2505.18668","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartgalaxy-a-dataset-for-infographic-chart#ran","syntology_url":"https://syntology.ai/paper/2505.18668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18668"}},"official":{"repos":["chartgalaxy/chartgalaxy"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-representation-intervention-really","slug":"does-representation-intervention-really","title":"Does Representation Intervention Really Identify Desired Concepts and Elicit Alignment?","date":"2025-05-24","arxiv_id":"2505.18672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-representation-intervention-really#ran","syntology_url":"https://syntology.ai/paper/2505.18672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18672"}},"official":null}},{"url":"/paper/mind-the-gap-a-practical-attack-on-gguf","slug":"mind-the-gap-a-practical-attack-on-gguf","title":"Mind the Gap: A Practical Attack on GGUF Quantization","date":"2025-05-24","arxiv_id":"2505.23786","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mind-the-gap-a-practical-attack-on-gguf#ran","syntology_url":"https://syntology.ai/paper/2505.23786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23786"}},"official":{"repos":["eth-sri/llm-quantization-attack"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/r-d-agent-quant-a-multi-agent-framework-for","slug":"r-d-agent-quant-a-multi-agent-framework-for","title":"R&D-Agent-Quant: A Multi-Agent Framework for Data-Centric Factors and Model Joint Optimization","date":"2025-05-21","arxiv_id":"2505.15155","repositories_listed":2,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/r-d-agent-quant-a-multi-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2505.15155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15155"}},"official":{"repos":["microsoft/rd-agent","microsoft/qlib"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/cad-coder-an-open-source-vision-language","slug":"cad-coder-an-open-source-vision-language","title":"CAD-Coder: An Open-Source Vision-Language Model for Computer-Aided Design Code Generation","date":"2025-05-20","arxiv_id":"2505.14646","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cad-coder-an-open-source-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.14646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14646"}},"official":{"repos":["anniedoris/cad-coder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omac-a-broad-optimization-framework-for-llm","slug":"omac-a-broad-optimization-framework-for-llm","title":"OMAC: A Broad Optimization Framework for LLM-Based Multi-Agent Collaboration","date":"2025-05-17","arxiv_id":"2505.11765","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/omac-a-broad-optimization-framework-for-llm#ran","syntology_url":"https://syntology.ai/paper/2505.11765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11765"}},"official":{"repos":["xiwenchao/omac-demo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/verithoughts-enabling-automated-verilog-code","slug":"verithoughts-enabling-automated-verilog-code","title":"VeriThoughts: Enabling Automated Verilog Code Generation using Reasoning and Formal Verification","date":"2025-05-16","arxiv_id":"2505.20302","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verithoughts-enabling-automated-verilog-code#ran","syntology_url":"https://syntology.ai/paper/2505.20302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20302"}},"official":{"repos":["wilyub/verithoughts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10607","slug":"2505-10607","title":"MONAQ: Multi-Objective Neural Architecture Querying for Time-Series Analysis on Resource-Constrained Devices","date":"2025-05-15","arxiv_id":"2505.10607","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2505-10607#ran","syntology_url":"https://syntology.ai/paper/2505.10607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10607"}},"official":{"repos":["kaist-dmlab/monaq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reviving-any-subset-autoregressive-models","slug":"reviving-any-subset-autoregressive-models","title":"Reviving Any-Subset Autoregressive Models with Principled Parallel Sampling and Speculative Decoding","date":"2025-04-29","arxiv_id":"2504.20456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reviving-any-subset-autoregressive-models#ran","syntology_url":"https://syntology.ai/paper/2504.20456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20456"}},"official":{"repos":["gabeguo/any-order-speculative-decoding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/osvbench-benchmarking-llms-on-specification","slug":"osvbench-benchmarking-llms-on-specification","title":"OSVBench: Benchmarking LLMs on Specification Generation Tasks for Operating System Verification","date":"2025-04-29","arxiv_id":"2504.20964","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osvbench-benchmarking-llms-on-specification#ran","syntology_url":"https://syntology.ai/paper/2504.20964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20964"}},"official":{"repos":["lishangyu-hkust/osvbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autop2c-an-llm-based-agent-framework-for-code","slug":"autop2c-an-llm-based-agent-framework-for-code","title":"AutoP2C: An LLM-Based Agent Framework for Code Repository Generation from Multimodal Content in Academic Papers","date":"2025-04-28","arxiv_id":"2504.20115","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autop2c-an-llm-based-agent-framework-for-code#ran","syntology_url":"https://syntology.ai/paper/2504.20115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20115"}},"official":{"repos":["shoushouyu/automated-paper-to-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/paper2code-automating-code-generation-from","slug":"paper2code-automating-code-generation-from","title":"Paper2Code: Automating Code Generation from Scientific Papers in Machine Learning","date":"2025-04-24","arxiv_id":"2504.17192","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paper2code-automating-code-generation-from#ran","syntology_url":"https://syntology.ai/paper/2504.17192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.17192"}},"official":{"repos":["going-doer/paper2code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/repost-scalable-repository-level-coding","slug":"repost-scalable-repository-level-coding","title":"RepoST: Scalable Repository-Level Coding Environment Construction with Sandbox Testing","date":"2025-03-10","arxiv_id":"2503.07358","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/repost-scalable-repository-level-coding#ran","syntology_url":"https://syntology.ai/paper/2503.07358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07358"}},"official":{"repos":["yiqingxyq/RepoST"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/fea-bench-a-benchmark-for-evaluating","slug":"fea-bench-a-benchmark-for-evaluating","title":"FEA-Bench: A Benchmark for Evaluating Repository-Level Code Generation for Feature Implementation","date":"2025-03-09","arxiv_id":"2503.06680","repositories_listed":0,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fea-bench-a-benchmark-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2503.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06680"}},"official":null}},{"url":"/paper/an-analyst-inspector-framework-for-evaluating","slug":"an-analyst-inspector-framework-for-evaluating","title":"An Analyst-Inspector Framework for Evaluating Reproducibility of LLMs in Data Science","date":"2025-02-23","arxiv_id":"2502.16395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-analyst-inspector-framework-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.16395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16395"}},"official":{"repos":["qunhualilab/llm-ds-reproducibility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/s-test-time-scaling-for-code-generation","slug":"s-test-time-scaling-for-code-generation","title":"S*: Test Time Scaling for Code Generation","date":"2025-02-20","arxiv_id":"2502.14382","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-test-time-scaling-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.14382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14382"}},"official":{"repos":["novasky-ai/skythought"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptivestep-automatically-dividing-reasoning","slug":"adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","arxiv_id":"2502.13943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptivestep-automatically-dividing-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.13943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13943"}},"official":{"repos":["lux0926/asprm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-agents-to-overcome-ambiguity-in","slug":"interactive-agents-to-overcome-ambiguity-in","title":"Interactive Agents to Overcome Ambiguity in Software Engineering","date":"2025-02-18","arxiv_id":"2502.13069","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-agents-to-overcome-ambiguity-in#ran","syntology_url":"https://syntology.ai/paper/2502.13069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13069"}},"official":{"repos":["sani903/interactivesweagents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-turn-by-turn-verifiers-for-dialogue","slug":"training-turn-by-turn-verifiers-for-dialogue","title":"Training Turn-by-Turn Verifiers for Dialogue Tutoring Agents: The Curious Case of LLMs as Your Coding Tutors","date":"2025-02-18","arxiv_id":"2502.13311","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-turn-by-turn-verifiers-for-dialogue#ran","syntology_url":"https://syntology.ai/paper/2502.13311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13311"}},"official":{"repos":["iwangjian/Coding-Tutor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gift-gibbs-fine-tuning-for-code-generation","slug":"gift-gibbs-fine-tuning-for-code-generation","title":"GiFT: Gibbs Fine-Tuning for Code Generation","date":"2025-02-17","arxiv_id":"2502.11466","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/gift-gibbs-fine-tuning-for-code-generation#ran","syntology_url":"https://syntology.ai/paper/2502.11466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11466"}},"official":{"repos":["alex-haochenli/gift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-fine-tuning-when-scaling-test-time","slug":"rethinking-fine-tuning-when-scaling-test-time","title":"Rethinking Fine-Tuning when Scaling Test-Time Compute: Limiting Confidence Improves Mathematical Reasoning","date":"2025-02-11","arxiv_id":"2502.07154","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-fine-tuning-when-scaling-test-time#ran","syntology_url":"https://syntology.ai/paper/2502.07154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07154"}},"official":{"repos":["allanraventos/refine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codesim-multi-agent-code-generation-and-1","slug":"codesim-multi-agent-code-generation-and-1","title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","date":"2025-02-08","arxiv_id":"2502.05664","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codesim-multi-agent-code-generation-and-1#ran","syntology_url":"https://syntology.ai/paper/2502.05664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05664"}},"official":null}},{"url":"/paper/the-elicitation-game-evaluating-capability","slug":"the-elicitation-game-evaluating-capability","title":"The Elicitation Game: Evaluating Capability Elicitation Techniques","date":"2025-02-04","arxiv_id":"2502.02180","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-elicitation-game-evaluating-capability#ran","syntology_url":"https://syntology.ai/paper/2502.02180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02180"}},"official":null}},{"url":"/paper/cweval-outcome-driven-evaluation-on","slug":"cweval-outcome-driven-evaluation-on","title":"CWEval: Outcome-driven Evaluation on Functionality and Security of LLM Code Generation","date":"2025-01-14","arxiv_id":"2501.08200","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cweval-outcome-driven-evaluation-on#ran","syntology_url":"https://syntology.ai/paper/2501.08200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08200"}},"official":{"repos":["co1lin/cweval"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/epicoder-encompassing-diversity-and","slug":"epicoder-encompassing-diversity-and","title":"EpiCoder: Encompassing Diversity and Complexity in Code Generation","date":"2025-01-08","arxiv_id":"2501.04694","repositories_listed":0,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":14,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/epicoder-encompassing-diversity-and#ran","syntology_url":"https://syntology.ai/paper/2501.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04694"}},"official":null}},{"url":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large","slug":"humaneval-pro-and-mbpp-pro-evaluating-large","title":"HumanEval Pro and MBPP Pro: Evaluating Large Language Models on Self-invoking Code Generation","date":"2024-12-30","arxiv_id":"2412.21199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2412.21199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21199"}},"official":{"repos":["CodeEval-Pro/CodeEval-Pro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seeker-towards-exception-safety-code","slug":"seeker-towards-exception-safety-code","title":"Seeker: Towards Exception Safety Code Generation with Intermediate Language Agents Framework","date":"2024-12-16","arxiv_id":"2412.11713","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeker-towards-exception-safety-code#ran","syntology_url":"https://syntology.ai/paper/2412.11713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11713"}},"official":{"repos":["XMZhangAI/Seeker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/commit0-library-generation-from-scratch","slug":"commit0-library-generation-from-scratch","title":"Commit0: Library Generation from Scratch","date":"2024-12-02","arxiv_id":"2412.01769","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commit0-library-generation-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2412.01769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01769"}},"official":{"repos":["commit-0/commit0"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cornstack-high-quality-contrastive-data-for","slug":"cornstack-high-quality-contrastive-data-for","title":"CoRNStack: High-Quality Contrastive Data for Better Code Retrieval and Reranking","date":"2024-12-01","arxiv_id":"2412.01007","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cornstack-high-quality-contrastive-data-for#ran","syntology_url":"https://syntology.ai/paper/2412.01007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01007"}},"official":{"repos":["gangiswag/cornstack"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-driven-programming-a-large-language","slug":"planning-driven-programming-a-large-language","title":"Planning-Driven Programming: A Large Language Model Programming Workflow","date":"2024-11-21","arxiv_id":"2411.14503","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-driven-programming-a-large-language#ran","syntology_url":"https://syntology.ai/paper/2411.14503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14503"}},"official":{"repos":["you68681/lpw"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prosec-fortifying-code-llms-with-proactive","slug":"prosec-fortifying-code-llms-with-proactive","title":"ProSec: Fortifying Code LLMs with Proactive Security Alignment","date":"2024-11-19","arxiv_id":"2411.12882","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prosec-fortifying-code-llms-with-proactive#ran","syntology_url":"https://syntology.ai/paper/2411.12882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12882"}},"official":{"repos":["PurCL/ProSec"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/squeezed-attention-accelerating-long-context","slug":"squeezed-attention-accelerating-long-context","title":"Squeezed Attention: Accelerating Long Context Length LLM Inference","date":"2024-11-14","arxiv_id":"2411.09688","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/squeezed-attention-accelerating-long-context#ran","syntology_url":"https://syntology.ai/paper/2411.09688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.09688"}},"official":{"repos":["SqueezeAILab/SqueezedAttention"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/suffixdecoding-a-model-free-approach-to","slug":"suffixdecoding-a-model-free-approach-to","title":"SuffixDecoding: Extreme Speculative Decoding for Emerging AI Applications","date":"2024-11-07","arxiv_id":"2411.04975","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/suffixdecoding-a-model-free-approach-to#ran","syntology_url":"https://syntology.ai/paper/2411.04975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04975"}},"official":{"repos":["snowflakedb/arcticinference"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/interaction2code-how-far-are-we-from","slug":"interaction2code-how-far-are-we-from","title":"Interaction2Code: Benchmarking MLLM-based Interactive Webpage Code Generation from Interactive Prototyping","date":"2024-11-05","arxiv_id":"2411.03292","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interaction2code-how-far-are-we-from#ran","syntology_url":"https://syntology.ai/paper/2411.03292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03292"}},"official":{"repos":["webpai/interaction2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metrex-a-benchmark-for-verilog-code-metric","slug":"metrex-a-benchmark-for-verilog-code-metric","title":"MetRex: A Benchmark for Verilog Code Metric Reasoning Using LLMs","date":"2024-11-05","arxiv_id":"2411.03471","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/metrex-a-benchmark-for-verilog-code-metric#ran","syntology_url":"https://syntology.ai/paper/2411.03471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03471"}},"official":{"repos":["scale-lab/MetRex"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/autovfx-physically-realistic-video-editing","slug":"autovfx-physically-realistic-video-editing","title":"AutoVFX: Physically Realistic Video Editing from Natural Language Instructions","date":"2024-11-04","arxiv_id":"2411.02394","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/autovfx-physically-realistic-video-editing#ran","syntology_url":"https://syntology.ai/paper/2411.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02394"}},"official":null}},{"url":"/paper/neo-saving-gpu-memory-crisis-with-cpu","slug":"neo-saving-gpu-memory-crisis-with-cpu","title":"NEO: Saving GPU Memory Crisis with CPU Offloading for Online LLM Inference","date":"2024-11-02","arxiv_id":"2411.01142","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/neo-saving-gpu-memory-crisis-with-cpu#ran","syntology_url":"https://syntology.ai/paper/2411.01142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01142"}},"official":null}},{"url":"/paper/selfcodealign-self-alignment-for-code","slug":"selfcodealign-self-alignment-for-code","title":"SelfCodeAlign: Self-Alignment for Code Generation","date":"2024-10-31","arxiv_id":"2410.24198","repositories_listed":2,"syntology":{"n":37,"n_ran":30,"n_constructed":3,"n_ran_checked":22,"n_instrument":8,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":0,"phrase":"30 ran (of which 3 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/selfcodealign-self-alignment-for-code#ran","syntology_url":"https://syntology.ai/paper/2410.24198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24198"}},"official":{"repos":["bigcode-project/selfcodealign"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["community","listed","official"]}}},{"url":"/paper/turtlebench-a-visual-programming-benchmark-in","slug":"turtlebench-a-visual-programming-benchmark-in","title":"TurtleBench: A Visual Programming Benchmark in Turtle Geometry","date":"2024-10-31","arxiv_id":"2411.00264","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turtlebench-a-visual-programming-benchmark-in#ran","syntology_url":"https://syntology.ai/paper/2411.00264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00264"}},"official":{"repos":["sinaris76/turtlebench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evocodebench-an-evolving-code-generation-1","slug":"evocodebench-an-evolving-code-generation-1","title":"EvoCodeBench: An Evolving Code Generation Benchmark with Domain-Specific Evaluations","date":"2024-10-30","arxiv_id":"2410.22821","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evocodebench-an-evolving-code-generation-1#ran","syntology_url":"https://syntology.ai/paper/2410.22821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22821"}},"official":null}},{"url":"/paper/scaling-llm-inference-with-optimized-sample","slug":"scaling-llm-inference-with-optimized-sample","title":"Scaling LLM Inference with Optimized Sample Compute Allocation","date":"2024-10-29","arxiv_id":"2410.22480","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-llm-inference-with-optimized-sample#ran","syntology_url":"https://syntology.ai/paper/2410.22480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22480"}},"official":{"repos":["leililab/osca"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-autoregression-fast-llms-via-self","slug":"beyond-autoregression-fast-llms-via-self","title":"Beyond Autoregression: Fast LLMs via Self-Distillation Through Time","date":"2024-10-28","arxiv_id":"2410.21035","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-autoregression-fast-llms-via-self#ran","syntology_url":"https://syntology.ai/paper/2410.21035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21035"}},"official":{"repos":["jdeschena/sdtt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llmopt-learning-to-define-and-solve-general","slug":"llmopt-learning-to-define-and-solve-general","title":"LLMOPT: Learning to Define and Solve General Optimization Problems from Scratch","date":"2024-10-17","arxiv_id":"2410.13213","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llmopt-learning-to-define-and-solve-general#ran","syntology_url":"https://syntology.ai/paper/2410.13213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13213"}},"official":{"repos":["caigaojiang/llmopt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-v-evaluating-visual-understanding","slug":"humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","arxiv_id":"2410.12381","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-v-evaluating-visual-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.12381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12381"}},"official":{"repos":["HumanEval-V/HumanEval-V-Benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effi-code-unleashing-code-efficiency-in","slug":"effi-code-unleashing-code-efficiency-in","title":"EffiCoder: Enhancing Code Generation in Large Language Models through Efficiency-Aware Fine-tuning","date":"2024-10-14","arxiv_id":"2410.10209","repositories_listed":2,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effi-code-unleashing-code-efficiency-in#ran","syntology_url":"https://syntology.ai/paper/2410.10209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10209"}},"official":{"repos":["huangd1999/effi-code","huangd1999/efficoder"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/itergen-iterative-structured-llm-generation","slug":"itergen-iterative-structured-llm-generation","title":"IterGen: Iterative Semantic-aware Structured LLM Generation with Backtracking","date":"2024-10-09","arxiv_id":"2410.07295","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/itergen-iterative-structured-llm-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07295"}},"official":{"repos":["uiuc-arc/itergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/da-code-agent-data-science-code-generation","slug":"da-code-agent-data-science-code-generation","title":"DA-Code: Agent Data Science Code Generation Benchmark for Large Language Models","date":"2024-10-09","arxiv_id":"2410.07331","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/da-code-agent-data-science-code-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07331"}},"official":null}},{"url":"/paper/learning-code-preference-via-synthetic","slug":"learning-code-preference-via-synthetic","title":"Learning Code Preference via Synthetic Evolution","date":"2024-10-04","arxiv_id":"2410.03837","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-code-preference-via-synthetic#ran","syntology_url":"https://syntology.ai/paper/2410.03837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03837"}},"official":{"repos":["amazon-science/llm-code-preference"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codejudge-evaluating-code-generation-with","slug":"codejudge-evaluating-code-generation-with","title":"CodeJudge: Evaluating Code Generation with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02184","repositories_listed":1,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":8,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/codejudge-evaluating-code-generation-with#ran","syntology_url":"https://syntology.ai/paper/2410.02184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02184"}},"official":{"repos":["VichyTong/CodeJudge"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/automl-agent-a-multi-agent-llm-framework-for","slug":"automl-agent-a-multi-agent-llm-framework-for","title":"AutoML-Agent: A Multi-Agent LLM Framework for Full-Pipeline AutoML","date":"2024-10-03","arxiv_id":"2410.02958","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automl-agent-a-multi-agent-llm-framework-for#ran","syntology_url":"https://syntology.ai/paper/2410.02958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02958"}},"official":{"repos":["DeepAuto-AI/automl-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/repograph-enhancing-ai-software-engineering","slug":"repograph-enhancing-ai-software-engineering","title":"RepoGraph: Enhancing AI Software Engineering with Repository-level Code Graph","date":"2024-10-03","arxiv_id":"2410.14684","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/repograph-enhancing-ai-software-engineering#ran","syntology_url":"https://syntology.ai/paper/2410.14684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14684"}},"official":{"repos":["ozyyshr/repograph"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-code-to-correctness-closing-the-last","slug":"from-code-to-correctness-closing-the-last","title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","date":"2024-10-02","arxiv_id":"2410.01215","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/from-code-to-correctness-closing-the-last#ran","syntology_url":"https://syntology.ai/paper/2410.01215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01215"}},"official":{"repos":["YerbaPage/MGDebugger"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codev-bench-how-do-llms-understand-developer","slug":"codev-bench-how-do-llms-understand-developer","title":"Codev-Bench: How Do LLMs Understand Developer-Centric Code Completion?","date":"2024-10-02","arxiv_id":"2410.01353","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codev-bench-how-do-llms-understand-developer#ran","syntology_url":"https://syntology.ai/paper/2410.01353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01353"}},"official":{"repos":["LingmaTongyi/Codev-Bench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-hallucinations-in-practical-code","slug":"llm-hallucinations-in-practical-code","title":"LLM Hallucinations in Practical Code Generation: Phenomena, Mechanism, and Mitigation","date":"2024-09-30","arxiv_id":"2409.20550","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-hallucinations-in-practical-code#ran","syntology_url":"https://syntology.ai/paper/2409.20550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20550"}},"official":{"repos":["deepsoftwareanalytics/llmcodinghallucination"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/moss-enabling-code-driven-evolution-and","slug":"moss-enabling-code-driven-evolution-and","title":"MOSS: Enabling Code-Driven Evolution and Context Management for AI Agents","date":"2024-09-24","arxiv_id":"2409.16120","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moss-enabling-code-driven-evolution-and#ran","syntology_url":"https://syntology.ai/paper/2409.16120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16120"}},"official":{"repos":["ghost-in-moss/ghostos"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rmcbench-benchmarking-large-language-models","slug":"rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","arxiv_id":"2409.15154","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rmcbench-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2409.15154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15154"}},"official":{"repos":["qing-yuan233/RMCBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/promsec-prompt-optimization-for-secure","slug":"promsec-prompt-optimization-for-secure","title":"PromSec: Prompt Optimization for Secure Generation of Functional Source Code with Large Language Models (LLMs)","date":"2024-09-19","arxiv_id":"2409.12699","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promsec-prompt-optimization-for-secure#ran","syntology_url":"https://syntology.ai/paper/2409.12699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12699"}},"official":{"repos":["mahmoudkanazzal/PromSec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-5-coder-technical-report","slug":"qwen2-5-coder-technical-report","title":"Qwen2.5-Coder Technical Report","date":"2024-09-18","arxiv_id":"2409.12186","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/qwen2-5-coder-technical-report#ran","syntology_url":"https://syntology.ai/paper/2409.12186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12186"}},"official":{"repos":["qwenlm/qwen2.5-coder"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autosafecoder-a-multi-agent-framework-for","slug":"autosafecoder-a-multi-agent-framework-for","title":"AutoSafeCoder: A Multi-Agent Framework for Securing LLM Code Generation through Static Analysis and Fuzz Testing","date":"2024-09-16","arxiv_id":"2409.10737","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/autosafecoder-a-multi-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.10737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10737"}},"official":{"repos":["secureaiautonomylab/autosafecoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reranking-laws-for-language-generation-a","slug":"reranking-laws-for-language-generation-a","title":"Reranking Laws for Language Generation: A Communication-Theoretic Perspective","date":"2024-09-11","arxiv_id":"2409.07131","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reranking-laws-for-language-generation-a#ran","syntology_url":"https://syntology.ai/paper/2409.07131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07131"}},"official":null}},{"url":"/paper/planning-in-natural-language-improves-llm","slug":"planning-in-natural-language-improves-llm","title":"Planning In Natural Language Improves LLM Search For Code Generation","date":"2024-09-05","arxiv_id":"2409.03733","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-in-natural-language-improves-llm#ran","syntology_url":"https://syntology.ai/paper/2409.03733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03733"}},"official":{"repos":["scaleapi/plansearch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/doce-finding-the-sweet-spot-for-execution","slug":"doce-finding-the-sweet-spot-for-execution","title":"DOCE: Finding the Sweet Spot for Execution-Based Code Generation","date":"2024-08-25","arxiv_id":"2408.13745","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doce-finding-the-sweet-spot-for-execution#ran","syntology_url":"https://syntology.ai/paper/2408.13745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13745"}},"official":{"repos":["deep-spin/doce"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-verilogeval-newer-llms-in-context","slug":"revisiting-verilogeval-newer-llms-in-context","title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","date":"2024-08-20","arxiv_id":"2408.11053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-verilogeval-newer-llms-in-context#ran","syntology_url":"https://syntology.ai/paper/2408.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11053"}},"official":{"repos":["nvlabs/verilog-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selective-prompt-anchoring-for-code","slug":"selective-prompt-anchoring-for-code","title":"Selective Prompt Anchoring for Code Generation","date":"2024-08-17","arxiv_id":"2408.09121","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/selective-prompt-anchoring-for-code#ran","syntology_url":"https://syntology.ai/paper/2408.09121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09121"}},"official":{"repos":["magic-yuantian/selective-prompt-anchoring"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/i-sheep-self-alignment-of-llm-from-scratch","slug":"i-sheep-self-alignment-of-llm-from-scratch","title":"I-SHEEP: Self-Alignment of LLM from Scratch through an Iterative Self-Enhancement Paradigm","date":"2024-08-15","arxiv_id":"2408.08072","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/i-sheep-self-alignment-of-llm-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2408.08072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08072"}},"official":{"repos":["multimodal-art-projection/I-SHEEP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00989","slug":"2408-00989","title":"On the Resilience of LLM-Based Multi-Agent Collaboration with Faulty Agents","date":"2024-08-02","arxiv_id":"2408.00989","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00989#ran","syntology_url":"https://syntology.ai/paper/2408.00989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00989"}},"official":{"repos":["cuhk-arise/mas-resilience"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/appworld-a-controllable-world-of-apps-and","slug":"appworld-a-controllable-world-of-apps-and","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","date":"2024-07-26","arxiv_id":"2407.18901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/appworld-a-controllable-world-of-apps-and#ran","syntology_url":"https://syntology.ai/paper/2407.18901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18901"}},"official":{"repos":["stonybrooknlp/appworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-pro-are-low-rank-adapters-properly","slug":"lora-pro-are-low-rank-adapters-properly","title":"LoRA-Pro: Are Low-Rank Adapters Properly Optimized?","date":"2024-07-25","arxiv_id":"2407.18242","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lora-pro-are-low-rank-adapters-properly#ran","syntology_url":"https://syntology.ai/paper/2407.18242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18242"}},"official":{"repos":["mrflogs/LoRA-Pro"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/origen-enhancing-rtl-code-generation-with","slug":"origen-enhancing-rtl-code-generation-with","title":"OriGen:Enhancing RTL Code Generation with Code-to-Code Augmentation and Self-Reflection","date":"2024-07-23","arxiv_id":"2407.16237","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/origen-enhancing-rtl-code-generation-with#ran","syntology_url":"https://syntology.ai/paper/2407.16237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16237"}},"official":{"repos":["pku-liang/origen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-for-verilog-generation","slug":"large-language-model-for-verilog-generation","title":"Large Language Model for Verilog Generation with Code-Structure-Guided Reinforcement Learning","date":"2024-07-21","arxiv_id":"2407.18271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-model-for-verilog-generation#ran","syntology_url":"https://syntology.ai/paper/2407.18271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18271"}},"official":{"repos":["CatIIIIIIII/veriseek"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ecco-can-we-improve-model-generated-code","slug":"ecco-can-we-improve-model-generated-code","title":"ECCO: Can We Improve Model-Generated Code Efficiency Without Sacrificing Functional Correctness?","date":"2024-07-19","arxiv_id":"2407.14044","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ecco-can-we-improve-model-generated-code#ran","syntology_url":"https://syntology.ai/paper/2407.14044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14044"}},"official":{"repos":["codeeff/ecco"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/beyond-correctness-benchmarking-multi","slug":"beyond-correctness-benchmarking-multi","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","date":"2024-07-16","arxiv_id":"2407.11470","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/beyond-correctness-benchmarking-multi#ran","syntology_url":"https://syntology.ai/paper/2407.11470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11470"}},"official":{"repos":["jszheng21/race"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/spider2-v-how-far-are-multimodal-agents-from","slug":"spider2-v-how-far-are-multimodal-agents-from","title":"Spider2-V: How Far Are Multimodal Agents From Automating Data Science and Engineering Workflows?","date":"2024-07-15","arxiv_id":"2407.10956","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spider2-v-how-far-are-multimodal-agents-from#ran","syntology_url":"https://syntology.ai/paper/2407.10956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10956"}},"official":{"repos":["xlang-ai/spider2-v"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-language-model-creativity-a-case#ran","syntology_url":"https://syntology.ai/paper/2407.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09007"}},"official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inversecoder-unleashing-the-power-of#ran","syntology_url":"https://syntology.ai/paper/2407.05700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05700"}},"official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/let-the-code-llm-edit-itself-when-you-edit","slug":"let-the-code-llm-edit-itself-when-you-edit","title":"Let the Code LLM Edit Itself When You Edit the Code","date":"2024-07-03","arxiv_id":"2407.03157","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/let-the-code-llm-edit-itself-when-you-edit#ran","syntology_url":"https://syntology.ai/paper/2407.03157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03157"}},"official":null}},{"url":"/paper/theoremllama-transforming-general-purpose","slug":"theoremllama-transforming-general-purpose","title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","date":"2024-07-03","arxiv_id":"2407.03203","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/theoremllama-transforming-general-purpose#ran","syntology_url":"https://syntology.ai/paper/2407.03203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03203"}},"official":{"repos":["RickySkywalker/TheoremLlama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discoverybench-towards-data-driven-discovery","slug":"discoverybench-towards-data-driven-discovery","title":"DiscoveryBench: Towards Data-Driven Discovery with Large Language Models","date":"2024-07-01","arxiv_id":"2407.01725","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discoverybench-towards-data-driven-discovery#ran","syntology_url":"https://syntology.ai/paper/2407.01725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01725"}},"official":{"repos":["allenai/discoverybench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/web2code-a-large-scale-webpage-to-code","slug":"web2code-a-large-scale-webpage-to-code","title":"Web2Code: A Large-scale Webpage-to-Code Dataset and Evaluation Framework for Multimodal LLMs","date":"2024-06-28","arxiv_id":"2406.20098","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web2code-a-large-scale-webpage-to-code#ran","syntology_url":"https://syntology.ai/paper/2406.20098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20098"}},"official":{"repos":["mbzuai-llm/web2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mpcoder-multi-user-personalized-code","slug":"mpcoder-multi-user-personalized-code","title":"MPCODER: Multi-user Personalized Code Generator with Explicit and Implicit Style Representation Learning","date":"2024-06-25","arxiv_id":"2406.17255","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mpcoder-multi-user-personalized-code#ran","syntology_url":"https://syntology.ai/paper/2406.17255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17255"}},"official":{"repos":["455849940/MPCoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/res-q-evaluating-code-editing-large-language","slug":"res-q-evaluating-code-editing-large-language","title":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","date":"2024-06-24","arxiv_id":"2406.16801","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/res-q-evaluating-code-editing-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16801"}},"official":{"repos":["qurrent-ai/res-q"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beear-embedding-based-adversarial-removal-of","slug":"beear-embedding-based-adversarial-removal-of","title":"BEEAR: Embedding-based Adversarial Removal of Safety Backdoors in Instruction-tuned Language Models","date":"2024-06-24","arxiv_id":"2406.17092","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beear-embedding-based-adversarial-removal-of#ran","syntology_url":"https://syntology.ai/paper/2406.17092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17092"}},"official":{"repos":["reds-lab/beear"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/indict-code-generation-with-internal","slug":"indict-code-generation-with-internal","title":"INDICT: Code Generation with Internal Dialogues of Critiques for Both Security and Helpfulness","date":"2024-06-23","arxiv_id":"2407.02518","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/indict-code-generation-with-internal#ran","syntology_url":"https://syntology.ai/paper/2407.02518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02518"}},"official":{"repos":["SalesforceAIResearch/indict_code_gen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bigcodebench-benchmarking-code-generation","slug":"bigcodebench-benchmarking-code-generation","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","date":"2024-06-22","arxiv_id":"2406.15877","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bigcodebench-benchmarking-code-generation#ran","syntology_url":"https://syntology.ai/paper/2406.15877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15877"}},"official":{"repos":["bigcode-project/bigcodebench","bigcode-project/bigcodebench-annotation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"eaaefe3b6d3c027ff0de86ff93799bd4ea717c878af50e7a941b3269a1728c5f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}