{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/humaneval/papers/ran/1","list_of":"/task/humaneval","task":"HumanEval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,76],"of":76,"counts":{"archive_papers_tagged":264,"with_a_code_link":135,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":264,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":65,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":65,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/humaneval/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/any4-learned-4-bit-numeric-representation-for","slug":"any4-learned-4-bit-numeric-representation-for","title":"any4: Learned 4-bit Numeric Representation for LLMs","date":"2025-07-07","arxiv_id":"2507.04610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/any4-learned-4-bit-numeric-representation-for#ran","syntology_url":"https://syntology.ai/paper/2507.04610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.04610"}},"official":{"repos":["facebookresearch/any4"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-for-speed-dilated-scheduling-for-masked","slug":"plan-for-speed-dilated-scheduling-for-masked","title":"Plan for Speed -- Dilated Scheduling for Masked Diffusion Language Models","date":"2025-06-23","arxiv_id":"2506.19037","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plan-for-speed-dilated-scheduling-for-masked#ran","syntology_url":"https://syntology.ai/paper/2506.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19037"}},"official":null}},{"url":"/paper/actor-critic-based-online-data-mixing-for","slug":"actor-critic-based-online-data-mixing-for","title":"Actor-Critic based Online Data Mixing For Language Model Pre-Training","date":"2025-05-29","arxiv_id":"2505.23878","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-critic-based-online-data-mixing-for#ran","syntology_url":"https://syntology.ai/paper/2505.23878","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23878"}},"official":null}},{"url":"/paper/rewriting-pre-training-data-boosts-llm","slug":"rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","arxiv_id":"2505.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rewriting-pre-training-data-boosts-llm#ran","syntology_url":"https://syntology.ai/paper/2505.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02881"}},"official":{"repos":["rioyokotalab/swallow-code-math"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/datadecide-how-to-predict-best-pretraining","slug":"datadecide-how-to-predict-best-pretraining","title":"DataDecide: How to Predict Best Pretraining Data with Small Experiments","date":"2025-04-15","arxiv_id":"2504.11393","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/datadecide-how-to-predict-best-pretraining#ran","syntology_url":"https://syntology.ai/paper/2504.11393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11393"}},"official":null}},{"url":"/paper/repost-scalable-repository-level-coding","slug":"repost-scalable-repository-level-coding","title":"RepoST: Scalable Repository-Level Coding Environment Construction with Sandbox Testing","date":"2025-03-10","arxiv_id":"2503.07358","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/repost-scalable-repository-level-coding#ran","syntology_url":"https://syntology.ai/paper/2503.07358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07358"}},"official":{"repos":["yiqingxyq/RepoST"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/kodcode-a-diverse-challenging-and-verifiable","slug":"kodcode-a-diverse-challenging-and-verifiable","title":"KodCode: A Diverse, Challenging, and Verifiable Synthetic Dataset for Coding","date":"2025-03-04","arxiv_id":"2503.02951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kodcode-a-diverse-challenging-and-verifiable#ran","syntology_url":"https://syntology.ai/paper/2503.02951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02951"}},"official":{"repos":["KodCode-AI/kodcode"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masrouter-learning-to-route-llms-for-multi","slug":"masrouter-learning-to-route-llms-for-multi","title":"MasRouter: Learning to Route LLMs for Multi-Agent Systems","date":"2025-02-16","arxiv_id":"2502.11133","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masrouter-learning-to-route-llms-for-multi#ran","syntology_url":"https://syntology.ai/paper/2502.11133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11133"}},"official":{"repos":["yanweiyue/masrouter"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codesim-multi-agent-code-generation-and-1","slug":"codesim-multi-agent-code-generation-and-1","title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","date":"2025-02-08","arxiv_id":"2502.05664","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codesim-multi-agent-code-generation-and-1#ran","syntology_url":"https://syntology.ai/paper/2502.05664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05664"}},"official":null}},{"url":"/paper/learning-to-generate-unit-tests-for-automated","slug":"learning-to-generate-unit-tests-for-automated","title":"Learning to Generate Unit Tests for Automated Debugging","date":"2025-02-03","arxiv_id":"2502.01619","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-generate-unit-tests-for-automated#ran","syntology_url":"https://syntology.ai/paper/2502.01619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01619"}},"official":{"repos":["archiki/utgendebug"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large","slug":"humaneval-pro-and-mbpp-pro-evaluating-large","title":"HumanEval Pro and MBPP Pro: Evaluating Large Language Models on Self-invoking Code Generation","date":"2024-12-30","arxiv_id":"2412.21199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-pro-and-mbpp-pro-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2412.21199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21199"}},"official":{"repos":["CodeEval-Pro/CodeEval-Pro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-scaling-scriptsize-mathtt-f-laws","slug":"inference-scaling-scriptsize-mathtt-f-laws","title":"Inference Scaling fLaws: The Limits of LLM Resampling with Imperfect Verifiers","date":"2024-11-26","arxiv_id":"2411.17501","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inference-scaling-scriptsize-mathtt-f-laws#ran","syntology_url":"https://syntology.ai/paper/2411.17501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17501"}},"official":{"repos":["benediktstroebl/inference-scaling-limits"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-driven-programming-a-large-language","slug":"planning-driven-programming-a-large-language","title":"Planning-Driven Programming: A Large Language Model Programming Workflow","date":"2024-11-21","arxiv_id":"2411.14503","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-driven-programming-a-large-language#ran","syntology_url":"https://syntology.ai/paper/2411.14503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14503"}},"official":{"repos":["you68681/lpw"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/perfcodegen-improving-performance-of-llm","slug":"perfcodegen-improving-performance-of-llm","title":"PerfCodeGen: Improving Performance of LLM Generated Code with Execution Feedback","date":"2024-11-18","arxiv_id":"2412.03578","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/perfcodegen-improving-performance-of-llm#ran","syntology_url":"https://syntology.ai/paper/2412.03578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03578"}},"official":{"repos":["SalesforceAIResearch/perfcodegen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/selfcodealign-self-alignment-for-code","slug":"selfcodealign-self-alignment-for-code","title":"SelfCodeAlign: Self-Alignment for Code Generation","date":"2024-10-31","arxiv_id":"2410.24198","repositories_listed":2,"syntology":{"n":37,"n_ran":30,"n_constructed":3,"n_ran_checked":22,"n_instrument":8,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":0,"phrase":"30 ran (of which 3 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/selfcodealign-self-alignment-for-code#ran","syntology_url":"https://syntology.ai/paper/2410.24198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24198"}},"official":{"repos":["bigcode-project/selfcodealign"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["community","listed","official"]}}},{"url":"/paper/humaneval-v-evaluating-visual-understanding","slug":"humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","arxiv_id":"2410.12381","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humaneval-v-evaluating-visual-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.12381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12381"}},"official":{"repos":["HumanEval-V/HumanEval-V-Benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/one-language-many-gaps-evaluating-dialect#ran","syntology_url":"https://syntology.ai/paper/2410.11005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11005"}},"official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/from-code-to-correctness-closing-the-last","slug":"from-code-to-correctness-closing-the-last","title":"From Code to Correctness: Closing the Last Mile of Code Generation with Hierarchical Debugging","date":"2024-10-02","arxiv_id":"2410.01215","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/from-code-to-correctness-closing-the-last#ran","syntology_url":"https://syntology.ai/paper/2410.01215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01215"}},"official":{"repos":["YerbaPage/MGDebugger"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-in-natural-language-improves-llm","slug":"planning-in-natural-language-improves-llm","title":"Planning In Natural Language Improves LLM Search For Code Generation","date":"2024-09-05","arxiv_id":"2409.03733","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-in-natural-language-improves-llm#ran","syntology_url":"https://syntology.ai/paper/2409.03733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03733"}},"official":{"repos":["scaleapi/plansearch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codexgraph-bridging-large-language-models-and","slug":"codexgraph-bridging-large-language-models-and","title":"CodexGraph: Bridging Large Language Models and Code Repositories via Code Graph Databases","date":"2024-08-07","arxiv_id":"2408.03910","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/codexgraph-bridging-large-language-models-and#ran","syntology_url":"https://syntology.ai/paper/2408.03910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03910"}},"official":{"repos":["modelscope/modelscope-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inversecoder-unleashing-the-power-of#ran","syntology_url":"https://syntology.ai/paper/2407.05700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05700"}},"official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/res-q-evaluating-code-editing-large-language","slug":"res-q-evaluating-code-editing-large-language","title":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","date":"2024-06-24","arxiv_id":"2406.16801","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/res-q-evaluating-code-editing-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16801"}},"official":{"repos":["qurrent-ai/res-q"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":21,"n_constructed":0,"n_ran_checked":20,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/chatglm-a-family-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12793"}},"official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semcoder-training-code-language-models-with","slug":"semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","arxiv_id":"2406.01006","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/semcoder-training-code-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2406.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01006"}},"official":{"repos":["arise-lab/semcoder"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-instruction-evolving-for-large","slug":"automatic-instruction-evolving-for-large","title":"Automatic Instruction Evolving for Large Language Models","date":"2024-06-02","arxiv_id":"2406.00770","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-instruction-evolving-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.00770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00770"}},"official":null}},{"url":"/paper/soap-enhancing-efficiency-of-generated-code","slug":"soap-enhancing-efficiency-of-generated-code","title":"EffiLearner: Enhancing Efficiency of Generated Code via Self-Optimization","date":"2024-05-24","arxiv_id":"2405.15189","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soap-enhancing-efficiency-of-generated-code#ran","syntology_url":"https://syntology.ai/paper/2405.15189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15189"}},"official":{"repos":["huangd1999/effilearner","huangd1999/soap"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-tuning-with-loss-over","slug":"instruction-tuning-with-loss-over","title":"Instruction Tuning With Loss Over Instructions","date":"2024-05-23","arxiv_id":"2405.14394","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/instruction-tuning-with-loss-over#ran","syntology_url":"https://syntology.ai/paper/2405.14394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14394"}},"official":{"repos":["zhengxiangshi/instructionmodelling"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unchosen-experts-can-contribute-too","slug":"unchosen-experts-can-contribute-too","title":"Unchosen Experts Can Contribute Too: Unleashing MoE Models' Power by Self-Contrast","date":"2024-05-23","arxiv_id":"2405.14507","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unchosen-experts-can-contribute-too#ran","syntology_url":"https://syntology.ai/paper/2405.14507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14507"}},"official":{"repos":["davidfanzz/scmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mapcoder-multi-agent-code-generation-for","slug":"mapcoder-multi-agent-code-generation-for","title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","date":"2024-05-18","arxiv_id":"2405.11403","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mapcoder-multi-agent-code-generation-for#ran","syntology_url":"https://syntology.ai/paper/2405.11403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11403"}},"official":{"repos":["md-ashraful-pramanik/mapcoder"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xft-unlocking-the-power-of-code-instruction","slug":"xft-unlocking-the-power-of-code-instruction","title":"XFT: Unlocking the Power of Code Instruction Tuning by Simply Merging Upcycled Mixture-of-Experts","date":"2024-04-23","arxiv_id":"2404.15247","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/xft-unlocking-the-power-of-code-instruction#ran","syntology_url":"https://syntology.ai/paper/2404.15247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15247"}},"official":{"repos":["ise-uiuc/xft"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/self-organized-agents-a-llm-multi-agent","slug":"self-organized-agents-a-llm-multi-agent","title":"Self-Organized Agents: A LLM Multi-Agent Framework toward Ultra Large-Scale Code Generation and Optimization","date":"2024-04-02","arxiv_id":"2404.02183","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-organized-agents-a-llm-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2404.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02183"}},"official":{"repos":["tsukushiai/self-organized-agent"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/top-leaderboard-ranking-top-coding","slug":"top-leaderboard-ranking-top-coding","title":"Top Leaderboard Ranking = Top Coding Proficiency, Always? EvoEval: Evolving Coding Benchmarks via LLM","date":"2024-03-28","arxiv_id":"2403.19114","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/top-leaderboard-ranking-top-coding#ran","syntology_url":"https://syntology.ai/paper/2403.19114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19114"}},"official":{"repos":["evo-eval/evoeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset","slug":"codeultrafeedback-an-llm-as-a-judge-dataset","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","date":"2024-03-14","arxiv_id":"2403.09032","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset#ran","syntology_url":"https://syntology.ai/paper/2403.09032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09032"}},"official":{"repos":["martin-wey/codeultrafeedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inficoder-eval-systematically-evaluating-the","slug":"inficoder-eval-systematically-evaluating-the","title":"InfiBench: Evaluating the Question-Answering Capabilities of Code Large Language Models","date":"2024-03-11","arxiv_id":"2404.07940","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inficoder-eval-systematically-evaluating-the#ran","syntology_url":"https://syntology.ai/paper/2404.07940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07940"}},"official":{"repos":["infi-coder/infibench-evaluation-harness","infi-coder/infibench-evaluator"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llm4decompile-decompiling-binary-code-with","slug":"llm4decompile-decompiling-binary-code-with","title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","date":"2024-03-08","arxiv_id":"2403.05286","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/llm4decompile-decompiling-binary-code-with#ran","syntology_url":"https://syntology.ai/paper/2403.05286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05286"}},"official":{"repos":["albertan017/LLM4Decompile"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ldb-a-large-language-model-debugger-via","slug":"ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","arxiv_id":"2402.16906","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ldb-a-large-language-model-debugger-via#ran","syntology_url":"https://syntology.ai/paper/2402.16906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16906"}},"official":{"repos":["floridsleeves/llmdebugger"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-or-memorization-data","slug":"generalization-or-memorization-data","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","date":"2024-02-24","arxiv_id":"2402.15938","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generalization-or-memorization-data#ran","syntology_url":"https://syntology.ai/paper/2402.15938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15938"}},"official":{"repos":["yihongdong/cdd-ted4llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opencodeinterpreter-integrating-code#ran","syntology_url":"https://syntology.ai/paper/2402.14658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14658"}},"official":null}},{"url":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/humaneval-on-latest-gpt-models-2024#ran","syntology_url":"https://syntology.ai/paper/2402.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14852"}},"official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generation-meets-verification-accelerating","slug":"generation-meets-verification-accelerating","title":"Generation Meets Verification: Accelerating Large Language Model Inference with Smart Parallel Auto-Correct Decoding","date":"2024-02-19","arxiv_id":"2402.11809","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generation-meets-verification-accelerating#ran","syntology_url":"https://syntology.ai/paper/2402.11809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11809"}},"official":{"repos":["cteant/space","hiyouga/llama-factory"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-evaluation-of-code-llms-with","slug":"unsupervised-evaluation-of-code-llms-with","title":"Unsupervised Evaluation of Code LLMs with Round-Trip Correctness","date":"2024-02-13","arxiv_id":"2402.08699","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-evaluation-of-code-llms-with#ran","syntology_url":"https://syntology.ai/paper/2402.08699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08699"}},"official":null}},{"url":"/paper/getting-the-most-out-of-your-tokenizer-for","slug":"getting-the-most-out-of-your-tokenizer-for","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","date":"2024-02-01","arxiv_id":"2402.01035","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/getting-the-most-out-of-your-tokenizer-for#ran","syntology_url":"https://syntology.ai/paper/2402.01035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01035"}},"official":{"repos":["gautierdag/tokenizer-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/oop-object-oriented-programming-evaluation","slug":"oop-object-oriented-programming-evaluation","title":"OOP: Object-Oriented Programming Evaluation Benchmark for Large Language Models","date":"2024-01-12","arxiv_id":"2401.06628","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/oop-object-oriented-programming-evaluation#ran","syntology_url":"https://syntology.ai/paper/2401.06628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06628"}},"official":{"repos":["alphadl/oop-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/repairllama-efficient-representations-and","slug":"repairllama-efficient-representations-and","title":"RepairLLaMA: Efficient Representations and Fine-Tuned Adapters for Program Repair","date":"2023-12-25","arxiv_id":"2312.15698","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/repairllama-efficient-representations-and#ran","syntology_url":"https://syntology.ai/paper/2312.15698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15698"}},"official":{"repos":["assert-kth/repairllama"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentcoder-multi-agent-based-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.13010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13010"}},"official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magicoder-source-code-is-all-you-need","slug":"magicoder-source-code-is-all-you-need","title":"Magicoder: Empowering Code Generation with OSS-Instruct","date":"2023-12-04","arxiv_id":"2312.02120","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/magicoder-source-code-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2312.02120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02120"}},"official":{"repos":["ise-uiuc/magicoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-benchmark-and-contamination-for#ran","syntology_url":"https://syntology.ai/paper/2311.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04850"}},"official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crosscodeeval-a-diverse-and-multilingual","slug":"crosscodeeval-a-diverse-and-multilingual","title":"CrossCodeEval: A Diverse and Multilingual Benchmark for Cross-File Code Completion","date":"2023-10-17","arxiv_id":"2310.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crosscodeeval-a-diverse-and-multilingual#ran","syntology_url":"https://syntology.ai/paper/2310.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11248"}},"official":null}},{"url":"/paper/codechain-towards-modular-code-generation","slug":"codechain-towards-modular-code-generation","title":"CodeChain: Towards Modular Code Generation Through Chain of Self-revisions with Representative Sub-modules","date":"2023-10-13","arxiv_id":"2310.08992","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codechain-towards-modular-code-generation#ran","syntology_url":"https://syntology.ai/paper/2310.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08992"}},"official":{"repos":["SalesforceAIResearch/CodeChain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-agent-tree-search-unifies-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.04406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04406"}},"official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/enhancing-large-language-models-in-coding#ran","syntology_url":"https://syntology.ai/paper/2309.17272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17272"}},"official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-channel-dimensions-to-isolate","slug":"rethinking-channel-dimensions-to-isolate","title":"Rethinking Channel Dimensions to Isolate Outliers for Low-bit Weight Quantization of Large Language Models","date":"2023-09-27","arxiv_id":"2309.15531","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-channel-dimensions-to-isolate#ran","syntology_url":"https://syntology.ai/paper/2309.15531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15531"}},"official":{"repos":["johnheo/adadim-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/code-llama-open-foundation-models-for-code","slug":"code-llama-open-foundation-models-for-code","title":"Code Llama: Open Foundation Models for Code","date":"2023-08-24","arxiv_id":"2308.12950","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/code-llama-open-foundation-models-for-code#ran","syntology_url":"https://syntology.ai/paper/2308.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12950"}},"official":{"repos":["facebookresearch/codellama"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/octopack-instruction-tuning-code-large","slug":"octopack-instruction-tuning-code-large","title":"OctoPack: Instruction Tuning Code Large Language Models","date":"2023-08-14","arxiv_id":"2308.07124","repositories_listed":3,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/octopack-instruction-tuning-code-large#ran","syntology_url":"https://syntology.ai/paper/2308.07124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07124"}},"official":{"repos":["bigcode-project/bigcode-evaluation-harness","bigcode-project/octopack"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/wizardcoder-empowering-code-large-language","slug":"wizardcoder-empowering-code-large-language","title":"WizardCoder: Empowering Code Large Language Models with Evol-Instruct","date":"2023-06-14","arxiv_id":"2306.08568","repositories_listed":4,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wizardcoder-empowering-code-large-language#ran","syntology_url":"https://syntology.ai/paper/2306.08568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08568"}},"official":{"repos":["nlpxucan/wizardlm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/large-language-models-of-code-fail-at","slug":"large-language-models-of-code-fail-at","title":"Large Language Models of Code Fail at Completing Code with Potential Bugs","date":"2023-06-06","arxiv_id":"2306.03438","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-of-code-fail-at#ran","syntology_url":"https://syntology.ai/paper/2306.03438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03438"}},"official":{"repos":["amazon-science/buggy-code-completion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leti-learning-to-generate-from-textual","slug":"leti-learning-to-generate-from-textual","title":"LeTI: Learning to Generate from Textual Interactions","date":"2023-05-17","arxiv_id":"2305.10314","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/leti-learning-to-generate-from-textual#ran","syntology_url":"https://syntology.ai/paper/2305.10314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10314"}},"official":{"repos":["xingyaoww/leti"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/codet5-open-code-large-language-models-for","slug":"codet5-open-code-large-language-models-for","title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","date":"2023-05-13","arxiv_id":"2305.07922","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/codet5-open-code-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2305.07922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07922"}},"official":{"repos":["salesforce/codet5"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/starcoder-may-the-source-be-with-you","slug":"starcoder-may-the-source-be-with-you","title":"StarCoder: may the source be with you!","date":"2023-05-09","arxiv_id":"2305.06161","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/starcoder-may-the-source-be-with-you#ran","syntology_url":"https://syntology.ai/paper/2305.06161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06161"}},"official":null}},{"url":"/paper/is-your-code-generated-by-chatgpt-really-1","slug":"is-your-code-generated-by-chatgpt-really-1","title":"Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation","date":"2023-05-02","arxiv_id":"2305.01210","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-your-code-generated-by-chatgpt-really-1#ran","syntology_url":"https://syntology.ai/paper/2305.01210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01210"}},"official":{"repos":["evalplus/evalplus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codegeex-a-pre-trained-model-for-code","slug":"codegeex-a-pre-trained-model-for-code","title":"CodeGeeX: A Pre-Trained Model for Code Generation with Multilingual Benchmarking on HumanEval-X","date":"2023-03-30","arxiv_id":"2303.17568","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/codegeex-a-pre-trained-model-for-code#ran","syntology_url":"https://syntology.ai/paper/2303.17568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17568"}},"official":{"repos":["THUDM/CodeGeeX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reflexion-language-agents-with-verbal","slug":"reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reflexion-language-agents-with-verbal#ran","syntology_url":"https://syntology.ai/paper/2303.11366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11366"}},"official":{"repos":["noahshinn024/reflexion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/recode-robustness-evaluation-of-code","slug":"recode-robustness-evaluation-of-code","title":"ReCode: Robustness Evaluation of Code Generation Models","date":"2022-12-20","arxiv_id":"2212.10264","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recode-robustness-evaluation-of-code#ran","syntology_url":"https://syntology.ai/paper/2212.10264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10264"}},"official":{"repos":["amazon-science/recode"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/parsel-a-unified-natural-language-framework","slug":"parsel-a-unified-natural-language-framework","title":"Parsel: Algorithmic Reasoning with Language Models by Composing Decompositions","date":"2022-12-20","arxiv_id":"2212.10561","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parsel-a-unified-natural-language-framework#ran","syntology_url":"https://syntology.ai/paper/2212.10561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10561"}},"official":{"repos":["ezelikman/parsel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-lingual-evaluation-of-code-generation","slug":"multi-lingual-evaluation-of-code-generation","title":"Multi-lingual Evaluation of Code Generation Models","date":"2022-10-26","arxiv_id":"2210.14868","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-lingual-evaluation-of-code-generation#ran","syntology_url":"https://syntology.ai/paper/2210.14868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14868"}},"official":{"repos":["amazon-research/mbxp-exec-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/contragen-effective-contrastive-learning-for","slug":"contragen-effective-contrastive-learning-for","title":"ContraCLM: Contrastive Learning For Causal Language Model","date":"2022-10-03","arxiv_id":"2210.01185","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/contragen-effective-contrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2210.01185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01185"}},"official":null}},{"url":"/paper/a-scalable-and-extensible-approach-to","slug":"a-scalable-and-extensible-approach-to","title":"MultiPL-E: A Scalable and Extensible Approach to Benchmarking Neural Code Generation","date":"2022-08-17","arxiv_id":"2208.08227","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-scalable-and-extensible-approach-to#ran","syntology_url":"https://syntology.ai/paper/2208.08227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.08227"}},"official":{"repos":["nuprl/multipl-e"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codet-code-generation-with-generated-tests","slug":"codet-code-generation-with-generated-tests","title":"CodeT: Code Generation with Generated Tests","date":"2022-07-21","arxiv_id":"2207.10397","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codet-code-generation-with-generated-tests#ran","syntology_url":"https://syntology.ai/paper/2207.10397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10397"}},"official":{"repos":["microsoft/codet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fault-aware-neural-code-rankers","slug":"fault-aware-neural-code-rankers","title":"Fault-Aware Neural Code Rankers","date":"2022-06-04","arxiv_id":"2206.03865","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fault-aware-neural-code-rankers#ran","syntology_url":"https://syntology.ai/paper/2206.03865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03865"}},"official":{"repos":["microsoft/coderanker"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-conversational-paradigm-for-program","slug":"a-conversational-paradigm-for-program","title":"CodeGen: An Open Large Language Model for Code with Multi-Turn Program Synthesis","date":"2022-03-25","arxiv_id":"2203.13474","repositories_listed":8,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-conversational-paradigm-for-program#ran","syntology_url":"https://syntology.ai/paper/2203.13474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13474"}},"official":{"repos":["salesforce/CodeGen","salesforce/jaxformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/evaluating-large-language-models-trained-on","slug":"evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","arxiv_id":"2107.03374","repositories_listed":13,"syntology":{"n":39,"n_ran":26,"n_constructed":0,"n_ran_checked":24,"n_instrument":2,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":23,"n_pointer_only":4,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 24 with no instrument failure: 1 honoured, 0 violated, 23 with no contract checked; 2 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/evaluating-large-language-models-trained-on#ran","syntology_url":"https://syntology.ai/paper/2107.03374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03374"}},"official":{"repos":["openai/human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["found_in_text","listed","official"]}}}],"record_sha256":"fa2d2425294d1d45759f7595b80550718374920445282dd5c70bf9c325fc7955","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}